<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e92727</article-id><article-id pub-id-type="doi">10.2196/92727</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Real-World Use of Controlled Terminologies, Ontologies, and Vocabularies for Evidence Generation Across a Large International Observational Network: Challenges and Lessons Learned From a Mixed Method Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Ostropolets</surname><given-names>Anna</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Korsik</surname><given-names>Vlad</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Skuhareuskaya</surname><given-names>Tatsiana</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhuk</surname><given-names>Aleh</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Khitrun</surname><given-names>Maryia</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Davydov</surname><given-names>Alexander</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Dymshyts</surname><given-names>Dmitry</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Reich</surname><given-names>Christian</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Hripcsak</surname><given-names>George</given-names></name><degrees>MD, MS</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Ryan</surname><given-names>Patrick</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib></contrib-group><aff id="aff1"><institution>Observational Health Data Analytics, Johnson &#x0026; Johnson</institution><addr-line>Raritan</addr-line><addr-line>NJ</addr-line><country>United States</country></aff><aff id="aff2"><institution>Department of Biomedical Informatics, Columbia University Irving Medical Center</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff3"><institution>Odysseus, an EPAM Company</institution><addr-line>Cambridge</addr-line><addr-line>MA</addr-line><country>United States</country></aff><aff id="aff4"><institution>Nemesis Health</institution><addr-line>New York</addr-line><addr-line>NY</addr-line><country>United States</country></aff><aff id="aff5"><institution>Observational Health Data Analytics, Johnson&#x0026;Johnson (United States)</institution><addr-line>Titusville</addr-line><addr-line>NJ</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>M&#x00E9;sz&#x00E1;ros</surname><given-names>&#x00C1;gota</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Groot</surname><given-names>Rowdy De</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Mea</surname><given-names>Vincenzo Della</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Park</surname><given-names>Yiju</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Anna Ostropolets, MD, PhD, Observational Health Data Analytics, Johnson &#x0026; Johnson, Raritan, NJ, 08869, United States; <email>ao2671@cumc.columbia.edu</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>15</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e92727</elocation-id><history><date date-type="received"><day>02</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>11</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>20</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Anna Ostropolets, Vlad Korsik, Tatsiana Skuhareuskaya, Aleh Zhuk, Maryia Khitrun, Alexander Davydov, Dmitry Dymshyts, Christian Reich, George Hripcsak, Patrick Ryan. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 15.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e92727"/><abstract><sec><title>Background</title><p>Large-scale international real-world evidence generation benefits from terminology harmonization. Despite widespread adoption of standardized vocabularies, their effective use and long-term sustainability at scale remain poorly understood.</p></sec><sec><title>Objective</title><p>This study aimed to examine real-world terminology and code use in data across a federated network of observational data sources.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a 2-part survey of researchers and data owners within the Observational Health Data Sciences and Informatics community on their terminology use, challenges, and needs, accompanied by the analysis of code use across a subset of real-world data sources.</p></sec><sec sec-type="results"><title>Results</title><p>The survey covered 144 institutions across the United States, the United Kingdom, Europe, Asia, and Africa. Data on terminology use covered 60 sources, including 22 data sources that provided detailed code-level use information. We observed significant variations in terminology use, with 61 out of 89 terminologies used in the data present in less than 10% of the data sources. Code use even after data harmonization was also highly variable: less than 1% (95/742,337) of codes were found in all data sources. Mapping and hierarchy completeness, terminology coverage, versioning, and terminology changes were among the most common challenges. We outlined several of our subsequent process improvements: community contribution and stewardship pipelines, metadata for relationships, and informatics tools for assessment of the impact of terminology change.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Terminology and coding inconsistencies across observational data sources require a standardized terminology system. Such a system is complex and time-consuming and needs community contribution and informatics solutions for harmonization to be scalable and sustainable. Even with a common reference standard, high heterogeneity of terminology and code use across different observational data sources remains.</p></sec></abstract><kwd-group><kwd>ontology</kwd><kwd>terminology</kwd><kwd>interoperability</kwd><kwd>observational studies</kwd><kwd>real-world data</kwd><kwd>real-world evidence</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Retrospective observational research using real-world observational data can provide insights into patient care and treatment outcomes across broad, diverse populations that may not be fully represented in randomized clinical trials [<xref ref-type="bibr" rid="ref1">1</xref>]. Observational data networks collect information from various real-world data sources, such as electronic health records (EHRs), administrative claims, and registries, and provide larger patient samples, which is particularly relevant for addressing complex clinical questions, including those related to rare diseases or new drugs [<xref ref-type="bibr" rid="ref2">2</xref>]. If networks are international, they can also capture care for diverse patient populations. On the other hand, the effective use of such networks for evidence generation requires reconciliation of differences across disparate ontologies, terminologies, and coding schemas to code health care events in different countries and institutions [<xref ref-type="bibr" rid="ref3">3</xref>]. Despite the long-standing use of medical terminologies, no studies have examined the real-world manifestation and use of ontologies in observational data sources at scale.</p><p>Differences in data collection and coding can be overcome by using Common Data Models (CDMs), data harmonization, and quality assurance processes. One approach, used in large networks, is to preserve source content and coding and reconcile differences at the analysis stage [<xref ref-type="bibr" rid="ref4">4</xref>]. Another approach, adopted by the Observational Health Data Sciences and Informatics (OHDSI) collaborative, is to use interoperable ontologies at the data conversion stage to enable standardized use of research tools [<xref ref-type="bibr" rid="ref5">5</xref>].</p><p>OHDSI encompasses 544 data sources harmonized to the Observational Medical Outcomes Partnership (OMOP) CDM from 54 countries [<xref ref-type="bibr" rid="ref6">6</xref>], including EHRs, administrative claims, registries, biobanks, and other data sources. OMOP CDM relies on a mandatory reference standard&#x2014;OHDSI Standardized Vocabularies. The system is a harmonized collection of ontologies, terminologies, and controlled vocabularies and is used by all data owners in the OHDSI network to code patient data in the data sources converted to OMOP CDM. The OHDSI Standardized Vocabularies aim to achieve (1) comprehensive coverage of clinical events in the network by hosting all relevant local terminologies and (2) interoperability by supporting crosswalks among terminologies within the Vocabularies [<xref ref-type="bibr" rid="ref5">5</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. They have one referent term or concept per semantic entity (called standard) and an equivalence relationship from other terms with the same meaning (nonstandard) to the standard. For example, data sources in the United States contain <italic>International Classification of Diseases, Tenth Revision, Clinical Modification</italic> (<italic>ICD10CM</italic>) diagnostic codes, the United Kingdom uses Read codes, Korea uses KCD7, and China uses <italic>International Classification of Diseases, Tenth Revision, Chinese Version</italic> (<italic>ICD10CN</italic>). Within OMOP CDM, they are mapped to the Systematized Nomenclature of Medicine Clinical Terms (SNOMED CT) so that all converted data sources have SNOMED CT codes in standard fields and source terminologies in source fields available for standardized analytics.</p><p>Despite the widespread adoption of OMOP CDM and its central reliance on the Standardized Vocabularies, there remains limited empirical research on how individual terminologies within the Vocabularies are used in practical research. Prior literature documented challenges in terminology systems, including incomplete domain coverage, inconsistencies in hierarchies, and redundancy [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. However, few studies have examined how these issues manifest in real-world, large-scale, multi-institutional research environments. It is also unclear how effectively a centralized terminology system can support diverse clinical data sources or how often data users encounter gaps that require manual curation or workarounds.</p><p>This study aims to fill that gap by systematically analyzing the use of terminologies and codes across real-world data sources within a global federated network. We combine quantitative analysis of terminology content and use patterns with qualitative insights from expert interviews to identify both patterns of use and persistent operational challenges. The study aims to (1) characterize the scale, complexity, and limitations of terminology integration in practical observational research and (2) outline methodological, process, and informatics solutions that are required to maintain a common terminology standard across disparate data sources.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>To assess the use of terminologies across the international network of observational data sources (hereon, data sources), the OHDSI Vocabulary team, which maintains the OHDSI Vocabularies, conducted a 6-month landscape assessment in 2023 that concurrently included (1) a 2-part survey distributed across the OHDSI community and (2) in-depth interviews with the members of the OHDSI community.</p><p>A survey was disseminated to the OHDSI community members through (1) the mailing list of individuals who had downloaded the OHDSI Vocabularies (&#x003E;8500 people); (2) continuous announcements on the OHDSI Microsoft Teams vocabulary channel (&#x003E;1400 members), forums, and community calls; and (3) outreach to the working group leads and individual community members.</p><p>The first part of the survey (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) targeted data owners and included questions about the data source to which they had access; terminologies, vocabularies, classifications, ontologies, and coding schemes used in their data; the frequency of data and Vocabularies refreshes; and the Vocabulary version in use. Hereon, the OHDSI Standardized Vocabularies system will be referred to as Vocabularies, and individual ontologies, terminologies, classifications, or vocabularies within the system as terminologies, regardless of the level of their formalization.</p><p>The second part of the landscape assessment survey (Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>) targeted data owners, data conversion specialists, researchers, and other users of the OHDSI Vocabularies. First, the Vocabulary team asked the responders to score their confidence in Vocabularies&#x2019; integrity on a 5-point Likert scale. Questions included the tasks they use Vocabularies for, perceived Vocabularies&#x2019; completeness, correctness, intuitiveness of use, Vocabularies-related issues, and areas for improvement. We then analyzed anonymized results of the survey obtained from the Vocabulary team using a combination of deductive and inductive thematic analysis. All challenges were identified and listed based on the responder input. The challenges were then grouped together to form themes, which were subsequently discussed and refined with a second investigator. Finally, themes were organized according to an adaptation of the existing data quality framework by Khan et al [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>The Vocabulary team concurrently used the convenience sampling method to select OHDSI open-source package and tool maintainers and leads of the working groups to conduct open-ended interviews. The interviews sought an in-depth understanding of the common challenges in using the Vocabularies within the OHDSI tools (CohortDiagnostics, FeatureExtraction, Atlas, and DataQualityDashboard) and their impact on clinical research (all clinical research groups: oncology and genomic, phenotyping, ophthalmology, dentistry, health equity, imaging, and psychiatry workgroups). During 1-hour semistructured online interviews conducted in January and February 2023, members of the Vocabulary team asked the interviewees to identify vocabulary dependencies in tools and research and provide an in-depth description of recent challenges they had experienced or that had been reported to them. At the end of the interview, key themes, challenges, and barriers were verbally summarized and validated with the interviewees, documented, and used to enrich the themes identified from the survey.</p><p>In addition, we asked data owners to supply aggregated code use information for their data sources. We collected standard and source codes along with their prevalence from the main OMOP CDM version 5 tables (condition_occurrence, procedure_occurrence, drug_exposure, device_exposure, measurement, observation). The executable SQL code distributed to data owners is available on the study package GitHub page [<xref ref-type="bibr" rid="ref13">13</xref>]. The output of the code comprised a list of concept identifiers from the OHDSI Standardized Vocabularies along with the associated frequency in the data. We aggregated the data across the data sources and analyzed and plotted the distribution of the unique and overlapping concepts.</p><p>Results of the landscape assessment informed the prioritization of systematic improvements to the content and processes of the Vocabularies. These improvements were initiated after the completion of the assessment and have continued from 2023 to the present. The Improvements section describes content improvement as well as process improvement (regular cadence, assessment of principles and changes, collaborative development, and community engagement) and their connection to the results of the assessment.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>The landscape assessment was reviewed by the Columbia University Institutional Review Board and was determined not to require further review.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>We first assessed real-world terminology and code use in data across the network.</p><sec id="s3-1"><title>Vocabulary Use in Source Data</title><p>The landscape assessment provided information on Vocabularies&#x2019; use across 60 data sources. Most of them originated in North America (United States: n=26, 43%, and Canada: n=3, 5%) and Europe (Belgium: n=5, 8%; Spain, Finland, Germany, Italy, and the United Kingdom: n=3, 5% each; the Netherlands: n=2, 3%; Norway, Estonia, Denmark, and Bulgaria: n=1, 2% each). Additional data sources originated from the Asia-Pacific region (Korea: n=2, 3%; and India and Japan: n=1, 2% each), as well as South America (Colombia: n=1, 2%). Most of the data were EHRs from tertiary health care centers and primary care centers (n=41, 68%), followed by administrative claims databases (n=9, 15%), registries (n=4, 7%), biobanks (n=3, 5%), clinical trial data (n=1, 2%), surveillance data (n=1, 2%), and combined EHR-registry dataset (n=1, 2%).</p><p>We observed high variability in vocabulary use across the data sources. While there were 12 terminologies used in at least one-third of the data sources, 16 additional terminologies were used in less than one-third but more than 10% of the data sources (such as CVX for vaccine capture; 11/61, 18%), and 60 other terminologies were used in less than 10% of the data sources (Table S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p><p>A total of 12 relatively common terminologies used in the data included the World Health Organization (WHO) and country-specific versions of the <italic>International Classification of Diseases, Tenth Revision</italic> (<italic>ICD-10</italic>) and <italic>International Classification of Diseases, Ninth Revision</italic> (<italic>ICD-9</italic>; n=50, 83% and n=44, 73%, respectively). Logical Observation Identifiers Names and Codes (LOINC), SNOMED CT, <italic>ICD-9 Procedure Coding System</italic> (<italic>ICD9Proc</italic>), and Current Procedural Terminology, Fourth Edition (CPT4) were used in more than half of the data sources, followed by the Anatomical Therapeutic Chemical (ATC) classification system, <italic>ICD-10 Procedure Coding System (ICD-10-PCS</italic>), Healthcare Common Procedure Coding System (HCPCS), RxNorm, National Drug Code (NDC), and <italic>International Classification of Diseases for Oncology, Third Revision</italic> (<italic>ICDO3</italic>) used in more than one-third of the data sources. <italic>ICD9Proc</italic>, CPT4, HCPCS, and <italic>ICD-10-PCS</italic> were used to code procedures and services; ATC, NDC, and RxNorm were used to code drugs; LOINC was used to code measurements and laboratory tests; and SNOMED CT was used to code for multiple domains. Use of SNOMED CT and LOINC included disparate local coding schemas (such as laboratory tests and flowsheets) that the data owners mapped to these terminologies.</p></sec><sec id="s3-2"><title>Code Use in Standardized Data</title><p>A subset of data sources also provided statistics on code use within their datasets. The aggregated dataset included code use data from 22 data sources, comprising 14 US and 8 non-US sources across administrative claims, EHRs, and hospital data (Table S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>), and comprised 742,337 unique standard concepts across 5 common OMOP CDM domains (Condition, Procedure, Measurement, Observation, and Drug). These data reflect code use following data standardization using the OHDSI Vocabularies.</p><p>Despite data harmonization, substantial variation in code use remained across data sources in the OHDSI network: 36% (270,452/742,337) of concepts were unique to a data source, 11% (n=78,608) concepts were shared across at least half of the data sources, and less than 1% (n=95) of concepts were shared across all data sources (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Distribution of concept counts across 22 data sources within the Observational Health Data Sciences and Informatics network, stratified by domain. The first bar represents the number of concepts found in 1 data source out of 22. Concepts found in more than 15 data sources (up to 1100) are not displayed for readability.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e92727_fig01.png"/></fig><p>Condition was the least heterogeneous domain, with the highest number of overlapping concepts across all domains, followed by procedures. Among all the domains, conditions underwent the longest and most extensive harmonization within the OHDSI Vocabularies. On the other hand, measurements lack established coding practices, explaining higher heterogeneity across domains. Drug coding variation could be due to different formulations in international markets.</p><p>Codes unique to a data source were not necessarily rare codes: some of the codes with a high number of patient records in data sources, such as laboratory tests, vitals, and scores, were only found in a few data sources (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><p><xref ref-type="table" rid="table1">Table 1</xref> shows an example of standard codes commonly used across the network to code body temperature (with or without measurement results) after data harmonization efforts. One rare code had &#x003C;2000 records across 8 data sources. On the contrary, some of the other more commonly used codes were found in 3 or fewer data sources: the most prevalent code &#x201C;Temperature&#x201D; was used only in 2 data sources, &#x201C;Oral temperature&#x201D; in 3 data sources, and &#x201C;Temperature of skin&#x201D; and &#x201C;Temperature taking&#x201D; in 1 data source.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Distribution of patient record counts per code across 22 data sources within the Observational Health Data Sciences and Informatics (OHDSI) network, colored by domain. Each dot represents a code in the OHDSI Standardized Vocabularies.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e92727_fig02.png"/></fig><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Selected measurement codes for body temperature measurement, with data source and patient record count across 22 data sources within the Observational Health Data Sciences and Informatics network.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Number of records</td><td align="left" valign="bottom">Number of databases</td><td align="left" valign="bottom">Concept name</td><td align="left" valign="bottom">Concept code</td><td align="left" valign="bottom">Vocabulary</td></tr></thead><tbody><tr><td align="left" valign="top">1,302,569,491</td><td align="left" valign="top">2</td><td align="left" valign="top">Temperature</td><td align="left" valign="top">703421000</td><td align="left" valign="top">SNOMED CT<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td></tr><tr><td align="left" valign="top">34,433,988</td><td align="left" valign="top">9</td><td align="left" valign="top">Body temperature</td><td align="left" valign="top">8310&#x2010;5</td><td align="left" valign="top">LOINC<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td></tr><tr><td align="left" valign="top">13,105,216</td><td align="left" valign="top">3</td><td align="left" valign="top">Oral temperature</td><td align="left" valign="top">8331&#x2010;1</td><td align="left" valign="top">LOINC</td></tr><tr><td align="left" valign="top">4,168,603</td><td align="left" valign="top">12</td><td align="left" valign="top">Vital signs (temperature, pulse, respiratory rate, and blood pressure) documented and reviewed (CAP) (EM)</td><td align="left" valign="top">2010F</td><td align="left" valign="top">CPT4<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">794,963</td><td align="left" valign="top">1</td><td align="left" valign="top">Temperature of skin</td><td align="left" valign="top">364537001</td><td align="left" valign="top">SNOMED CT</td></tr><tr><td align="left" valign="top">462,982</td><td align="left" valign="top">1</td><td align="left" valign="top">Temperature taking</td><td align="left" valign="top">56342008</td><td align="left" valign="top">SNOMED CT</td></tr><tr><td align="left" valign="top">1343</td><td align="left" valign="top">8</td><td align="left" valign="top">Monitoring of Temperature External Approach</td><td align="left" valign="top">4A1ZXKZ</td><td align="left" valign="top"><italic>ICD-10-PCS</italic><sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>SNOMED CT: Systematized Nomenclature of Medicine Clinical Terms.</p></fn><fn id="table1fn2"><p><sup>b</sup>LOINC: Logical Observation Identifiers Names and Codes.</p></fn><fn id="table1fn3"><p><sup>c</sup>CPT4: Current Procedural Terminology, Fourth Edition.</p></fn><fn id="table1fn4"><p><sup>d</sup><italic>ICD-10-PCS</italic>: <italic>International Classification of Diseases, Tenth Revision, Procedure Coding System.</italic></p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Vocabularies&#x2019; Acceptance and Challenges</title><p>We then assessed how well Vocabularies facilitate data harmonization and standardized analytics across the network. The second part of the landscape assessment received 183 responses from community members from 144 institutions across the United States, the United Kingdom, Europe, Asia, and Africa. Overall, 87% (159/183) of the community were confident about the Vocabularies&#x2019; integrity, with more than one-quarter of responders (n=49, 27%) feeling extremely confident.</p><p>A substantial part of the community reported using the Vocabularies for more than 1 task. Of the 183 respondents, 143 (78%) use the Vocabularies for transforming data into the OMOP CDM; 119 (65%) use the Vocabularies for research (with characterization being the most common study type); and 51 (28%) use the Vocabularies for software, tools, and method development, including standardized analytics, data transformation, natural language processing, clinical decision support, and knowledge engineering.</p><p>On the basis of the results of the thematic analysis, community-identified challenges and areas for further development related to the Vocabularies were organized into 3 categories: conformance, completeness, and recency (<xref ref-type="table" rid="table2">Table 2</xref>) [<xref ref-type="bibr" rid="ref12">12</xref>]. Within the conformance category, limited mapping completeness and multiple standard concepts per semantic meaning were the most reported problems that contributed to lower confidence in the Vocabularies&#x2019; integrity. Specifically, the lack of mappings for nonstandard source concepts was a particular issue for European data sources. While source codes were available for querying in OMOP CDM, the lack of mappings to their standard counterparts in the common reference standard limited the use of standardized OHDSI tools that rely on the latter.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Main challenges with the Vocabularies and potential solutions.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Challenges</td><td align="left" valign="bottom">Example</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="2">Conformance</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Erroneous mappings from source to standard<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>M08.00 &#x201C;Unspecified juvenile rheumatoid arthritis of unspecified site&#x201D; is now mapped to &#x201C;Polyarticular juvenile idiopathic arthritis&#x201D; but should be mapped to &#x201C;Juvenile idiopathic arthritis&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Mappings associated with information loss<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Upward (uphill) mappings: Y82.8 &#x201C;Other medical devices associated with adverse incidents&#x201D; is mapped to &#x201C;Medical accident to patient during surgical and medical care&#x201D;</p></list-item><list-item><p>One: many mappings: O33.7 &#x201C;Maternal care for disproportion due to other fetal deformities&#x201D; is mapped to &#x201C;Fetal condition affecting obstetrical care of mother&#x201D; and &#x201C;Fetal disproportion&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Multiple standard concepts</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>SNOMED CT<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> 23986001 Condition &#x201C;Glaucoma&#x201D; and UK Biobank 100523&#x2010;2 survey answer &#x201C;Glaucoma&#x201D; (question: &#x201C;Eye problems/disorders&#x201D;)</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Completeness</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Insufficient mapping coverage</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>CDISC code C199995 &#x201C;/100x FIELD&#x201D; is not mapped</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Lack of comprehensive hierarchies</td><td align="left" valign="top"><list list-type="bullet"><list-item><p><italic>ICD-10-PCS</italic><sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup> 06LY7ZC &#x201C;Occlusion of Hemorrhoidal Plexus Via Natural or Artificial Opening&#x201D; is not hierarchically related to CPT4 1007789 &#x201C;Hemorrhoidectomy, internal, by ligation other than rubber band&#x201D;</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Source vocabularies not included in the Vocabularies</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>RadLex and radiology AGFA Impax are not incorporated into the Vocabularies</p></list-item></list></td></tr><tr><td align="left" valign="top" colspan="2">Recency and versioning</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Variability in release cadence</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Variable number of releases per year based on community requests</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Maintenance of one version of the Vocabularies available for download</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Only the current version of the Vocabularies is available for download on [<xref ref-type="bibr" rid="ref14">14</xref>]</p></list-item></list></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Unpredicted impact of the Vocabularies&#x2019; changes on ETL<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup> processes, research, and OHDSI<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup> tools</td><td align="left" valign="top"><list list-type="bullet"><list-item><p>Changes in the SNOMED CT hierarchy result in different sets of codes across different versions</p></list-item><list-item><p>Changes in domain assignment across versions impact table assignment during ETL</p></list-item></list></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Source codes are available for querying in the Observational Medical Outcomes Partnership Common Data Model.</p></fn><fn id="table2fn2"><p><sup>b</sup>SNOMED CT: Systematized Nomenclature of Medicine Clinical Terms.</p></fn><fn id="table2fn3"><p><sup>c</sup><italic>ICD-10-PCS</italic>: <italic>International Classification of Diseases, Tenth Revision, Procedure Coding System.</italic></p></fn><fn id="table2fn4"><p><sup>d</sup>ELT: extract, transform, and load.</p></fn><fn id="table2fn5"><p><sup>e</sup>OHDSI: Observational Health Data Sciences and Informatics. </p></fn></table-wrap-foot></table-wrap><p>The presence of multiple standard concepts and the lack of comprehensive hierarchies posed challenges for defining sets of codes to identify patients of interest (phenotyping). A single semantic concept, such as a specific disease, may be represented by different codes in different countries and mapped to multiple standard concepts within the Vocabularies. Researchers reported difficulties identifying all relevant concepts within the Vocabularies to ensure that concept sets comprehensively represented the intended clinical concept and captured all eligible patients.</p><p>Within the completeness category, source terminologies not yet incorporated into the OHDSI Vocabularies were among the most reported problems. These terminologies could be classified into two categories: (1) public terminologies maintained by specialized authoring organizations and (2) unstructured elements present in the data as source coding schemes or free text. Public terminologies were European, such as local flavors of <italic>ICD-10</italic> or International Classification of Primary Care, which is associated with a recent uptake of newly harmonized data sources in the European Health Data and Evidence Network [<xref ref-type="bibr" rid="ref15">15</xref>], as well as those covering new areas of interest in observational research (eg, RadLex and Radiology AGFA Impax for imaging data). Unstructured data source&#x2013;specific elements not covered by the Vocabularies included poorly structured data elements from surveys and questionnaires, EHR flowsheets, and local codes used in registries.</p><p>Within the recency and versioning category, assessment of the impact of terminology changes on common research tasks, such as phenotyping, was one of the main challenges. Community members reported that changes across SNOMED CT hierarchy lead to changes in concept sets created on different versions of the Vocabularies. Similarly, differences in domain assignment across different versions resulted in the placement of the same codes in different OMOP CDM tables during data conversion, affecting their subsequent retrieval. This issue had not been notable prior to the landscape assessment, as its impact on study design and execution became apparent only as the number of studies using OMOP CDM increased and the effects of changes in source and standard vocabularies became more evident.</p></sec><sec id="s3-4"><title>Improvements</title><sec id="s3-4-1"><title>Overview</title><p>First, terminology use across the network informed the prioritization of terminologies for maintenance and improvement. This prioritization was undertaken by a newly established Vocabulary Committee, comprising diverse stakeholders from across the community and serving as a governance body. The process resulted in a roadmap and release schedule [<xref ref-type="bibr" rid="ref16">16</xref>].</p></sec><sec id="s3-4-2"><title>Recency and Versioning</title><p>Second, to address the observed variability in release cadence, we introduced a fixed semiannual cadence of releases, with releases in August and February every year aligned with the release cadence of source terminologies. The OHDSI Standardized Vocabularies had previously followed a flexible release cycle based on user requests and needs. It was hypothesized that semiannual releases would promote greater consistency in the Vocabularies&#x2019; version across the network, thereby facilitating more consistent research. Communication around releases was further formalized to make Vocabularies&#x2019; changes more predictable for the community. Specifically, advance notifications of upcoming changes [<xref ref-type="bibr" rid="ref17">17</xref>] and detailed release notes [<xref ref-type="bibr" rid="ref18">18</xref>] were introduced, incorporating plain language summaries of their potential impact on research and extract, transform, and load processes.</p><p>Semiannual releases enabled greater focus on content development addressing terminology gaps identified by the landscape assessment. For example, responders noted unstable domain assignment as disruptive to efficient phenotyping, which resulted in a dedicated release targeted at improving SNOMED CT processing and harmonization.</p><p>Additionally, unpredicted impact of Vocabularies&#x2019; changes was partially subsequently addressed by community-developed tools that assess the impact of Vocabularies&#x2019; changes on phenotyping [<xref ref-type="bibr" rid="ref19">19</xref>] to allow the users to adjust cohort definitions based on changes in reference hierarchies and mappings and maintain concept sets consistent with the original logic.</p><p>Finally, to address the lack of versioning, we launched a versioning feature on Athena, OHDSI&#x2019;s Vocabularies distribution service, in August 2024, which allows users to download different vocabulary versions and assess the delta between 2 given releases.</p></sec><sec id="s3-4-3"><title>Completeness and Conformance</title><p>Besides focusing on the most used terminologies in the road map, we had to address the need for more terminologies and more mappings. Given that the demand substantially exceeded potential capacity of the road map, we developed a semiautomated pipeline to lower barriers to community engagement [<xref ref-type="bibr" rid="ref20">20</xref>]. This pipeline included templates supporting the most commonly reported community use cases: seamless incorporation of new source terminologies into the Vocabularies, adding or changing mappings, and adding or changing concept attributes (community contribution GitHub page [<xref ref-type="bibr" rid="ref20">20</xref>], part 1). These use cases could be handled through Microsoft Excel&#x2013;based templates and were sufficiently covered by quality assurance checks to enable rapid integration into the subsequent vocabulary release cycle, thereby reducing the latency between identification of an issue and its resolution. Throughout the 4 releases in 2024 and 2025, we incorporated 17 contributions: 8 with mapping changes for 1 or more relationships, 4 with concept attribute changes, 3 deduplications of 1 or more standard concept pairs, and 2 with new source terminologies.</p><p>We also partially enabled more complex contributions, such as additions of complex standard terminologies, semantic deduplication, and hierarchy construction, although these activities require more complex, often expert-driven quality assurance. Over the past several releases, we processed 6 complex contributions, including updates to complex terminologies impacting reference sets, such as the Korean Electronic Data Interchange code system or the Columbia International eHealth Laboratory; addition of new drug terminologies; and development of new drug route hierarchies. The number of contributions grew over time and required extensive documentation on setting up a local instance of the Vocabularies&#x2019; schema so that contributors can perform terminology development, testing, and integration before contribution submission (community contribution GitHub page, part 2). Similarly, in response to the growing need for maintenance and improvement of existing terminologies, we started decentralization of terminology maintenance through the introduction of terminology stewards&#x2014;individuals and institutions responsible for the maintenance of terminologies they use or are interested in.</p><p>Conformance challenges were addressed in several ways.</p><p>First, mapping errors have been addressed through targeted work on source terminologies within the semiannual release cycle, as well as enabling users to adjust mappings through community contributions.</p><p>Second, mapping with loss of information was addressed through introducing Simple Standard for Sharing Ontological Mappings (SSSOM) metadata for community contributions and for Vocabularies&#x2019; relationships. Metadata details include the mapping source, tool, confidence score, mapper, reviewer, and mapping predicate or precision. We curated SSSOM-compatible metadata for 142,113 mapping relationships [<xref ref-type="bibr" rid="ref21">21</xref>]. Mapping precision has 3 categories: broadMatch (mapping with loss of granularity, 17% of the annotated mappings), narrowMatch (mapping with increased granularity, &#x003C;1%), and exactMatch (83%). Metadata for concepts were also curated and distributed, including categories of nonstandard codes, as described in the Vocabularies&#x2019; interoperability documentation, and a reuse flag identifying codes reused across source terminologies [<xref ref-type="bibr" rid="ref18">18</xref>]. Concept and relationship metadata were made publicly available and accompany each release [<xref ref-type="bibr" rid="ref22">22</xref>].</p><p>Finally, together with the community, we have been addressing the challenge of multiple standard concepts. This challenge is most prominent in the areas where no single reference standard exists, and more terminology development work is warranted. Examples include harmonization of surveys and patient-reported outcomes, vaccine [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>] and device [<xref ref-type="bibr" rid="ref26">26</xref>] harmonization, development of ontologies to support observational research in Africa [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>], and laboratory test harmonization. Complexity of harmonization pointed to the need for collaboration with other communities, such as the LOINC-SNOMED CT collaboration [<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This work is the first study to examine real-world terminology use for observational research at scale. Our assessment of data sources from the United States, Europe, and the Asia-Pacific region revealed substantial heterogeneity of terminologies used in the data, with hundreds of terminologies, ontologies, and coding schemas used even within a limited subset of the data sources harmonized to OMOP CDM. Such diversity highlights the critical importance of ontology harmonization, as no single study protocol could feasibly account for the vast array of relevant codes across such a wide range of terminologies. Modern analytic techniques, such as the large-scale propensity score and predictive modeling, require far too many clinical variables as inputs to expect each data source owner to map all required source elements to concepts defined in a study. In this regard, the OMOP CDM, alongside the OHDSI Standardized Vocabularies, offers a unique advantage for generating generalizable and reproducible evidence by harmonizing both schema and content, as opposed to systems that rely solely on &#x201C;adaptive rules&#x201D; and keep source content inconsistent [<xref ref-type="bibr" rid="ref4">4</xref>]. On the other hand, the approach of the common reference standard has been proven to be resource-consuming. OHDSI best practices require aligning terminologies to ensure a single standard concept per semantic meaning and constructing comprehensive hierarchies. As shown in the Results section, such an endeavor takes years and requires sustainable processes, community engagement, and collaboration with internal and external partners. These findings are consistent with the experience of other terminology organizations. For example, the SNOMED CT-LOINC collaboration, which focuses on harmonizing laboratory test concepts, was formally launched in 2022, with the first production release in 2025; harmonization efforts have continued since then.</p><p>Diversity in code use across the network also underscores a less-discussed challenge: the portability of phenotype or cohort definitions. When only a limited set of codes is shared across data sources, and the majority are unique to specific sources, a phenotype with high performance on one data source may not yield similar results on another. There is only limited research on patterns of phenotype portability [<xref ref-type="bibr" rid="ref30">30</xref>-<xref ref-type="bibr" rid="ref32">32</xref>], and borrowing phenotypes from the literature without adequate validation remains a common practice [<xref ref-type="bibr" rid="ref33">33</xref>]. Although heterogeneity in coding practices has been highlighted before, researchers still rely on code sets developed in an institution to represent the same set of patients in their data [<xref ref-type="bibr" rid="ref34">34</xref>]. Our findings further strengthen previous research and amplify the magnitude of observed heterogeneity across disparate data sources [<xref ref-type="bibr" rid="ref35">35</xref>,<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>Beyond coding heterogeneity, changes in terminologies driven by an evolving understanding of diseases and quality improvements create additional challenges in ensuring temporal portability. A classic example of differences between <italic>ICD-9</italic> and <italic>ICD-10</italic>, which were reconciled through a CMS-curated crosswalk, is still being evaluated for validity in certain areas 8 years after the transition [<xref ref-type="bibr" rid="ref37">37</xref>]. With dozens of terminologies in use in the network, harmonization becomes increasingly complex, as reflected in the results of the landscape assessment. The latter showed that erroneous mappings, mappings with loss of information, and imperfect hierarchies are among the common problems in terminology harmonization. These findings strengthen previous research showing persistent challenges in the alignment of commonly used terminologies [<xref ref-type="bibr" rid="ref9">9</xref>-<xref ref-type="bibr" rid="ref11">11</xref>,<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Previous literature has mostly discussed the alignment of 2 given terminologies, whereas our experience highlights that the task becomes increasingly complex when aligning multiple terminologies.</p><p>We hypothesized that the solution to the scalability problem was to establish pipelines to enable people with different backgrounds from different countries to contribute to the Vocabularies&#x2019; content. Besides having domain expertise, we observed that community contributors had to learn about processes and operations within the OHDSI Vocabularies. Despite expanded documentation, tutorials, and other means of communication, the same questions continued to appear frequently in forums and working groups, suggesting that static documentation alone may not be sufficient to meet users&#x2019; learning needs. Previous research also highlights the importance of domain expertise in crowdsourcing [<xref ref-type="bibr" rid="ref40">40</xref>] and suggests objective prequalification assessment as a means to improve collaborative authoring [<xref ref-type="bibr" rid="ref41">41</xref>].</p><p>While our pipelines and documentation have supported community contributions, there remains significant potential for further development. New research domains within OHDSI&#x2014;such as imaging, dentistry, and surveys&#x2014;also necessitate both the addition of new content and the creation of conceptual models for content representation and quality assurance procedures to assess such representations, which delays implementation. Our experience with the contributions mirrors previously reported experience of other terminology systems, such as SNOMED, showing that contributions may introduce quality issues and require extensive quality assurance and curation [<xref ref-type="bibr" rid="ref42">42</xref>,<xref ref-type="bibr" rid="ref43">43</xref>], and points to the need for better informatics solutions. This observation is further supported by the growing body of literature examining the use of large language models and mixed approaches for terminology harmonization, including broad research and specific applications within OHDSI [<xref ref-type="bibr" rid="ref44">44</xref>-<xref ref-type="bibr" rid="ref48">48</xref>].</p><p>Overall, further work is needed to lower the barriers to community engagement and improve scalability and robustness, ensuring that a system of terminologies serving as the common reference standard can meet evidence-generation needs at scale.</p></sec><sec id="s4-2"><title>Limitations</title><p>This work has several limitations. First, as participation in the assessment was voluntary, not all the community members filled out the survey or contributed the data, which may limit the generalizability of the findings. Furthermore, because the total size and composition of the community are not precisely known, we are unable to reliably estimate the coverage of the survey responses. As such, the findings derived from survey data may reflect the perspectives of more engaged or vocal community members rather than the broader user base. Similarly, code use was assessed in the data sources that supplied information from 2 main regions, North America and Europe. Therefore, our findings may not be generalizable to other regions.</p></sec><sec id="s4-3"><title>Conclusions</title><p>While the use of a common reference standard, such as OHDSI Standardized Vocabularies, facilitates consistency in international observational research, substantial heterogeneity in terminology and code use across data sources remains even after harmonization. Efficient network research should account for such heterogeneity by developing tools and methods for robust phenotyping and making further efforts to improve Vocabularies&#x2019; coverage and quality. As the development and maintenance of such a system are complex, time-consuming endeavors, community input has played a central role in identifying high-priority content areas, surfacing usability issues, and revealing process inefficiencies. In parallel, content contributions from the community can be a sustainable approach to cover new use cases, increase the intake of new terminologies, and crowdsource improvement activities.</p></sec></sec></body><back><ack><p>The authors would like to thank the Observational Health Data Sciences and Informatics (OHDSI) community for contributing to the content of the OHDSI Vocabularies and engaging in scientific discourse to improve their content and processes. Large language models (Claude Sonnet 4.5) were used to suggest language improvements to the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was partially supported by National Institutes of Health grant R01 LM006910 and Chan Zuckerberg Initiative "Open Source Ontologies to Power an Open Science Community" grant 2022-309701.</p></sec><sec><title>Data Availability</title><p>Observational Health Data Sciences and Informatics (OHDSI) Standardized Vocabularies are open source and publicly available at Athena [<xref ref-type="bibr" rid="ref49">49</xref>] and GitHub [<xref ref-type="bibr" rid="ref18">18</xref>]. Most vocabularies can be downloaded for free. Vocabularies requiring an end user license agreement are distributed upon proof of license with the authoring organization. Aggregated concept counts are available at Atlas [<xref ref-type="bibr" rid="ref50">50</xref>] under Network Prevalence Count.</p></sec></notes><fn-group><fn fn-type="con"><p>CR, GH, and PR contributed equally to this work as senior authors. CR, GH, PR, and AO conceptualized the study design. AO drafted the manuscript. All authors contributed to data acquisition, analysis, and tool development, and reviewed the manuscript.</p></fn><fn fn-type="conflict"><p>AO, DD, and PR are employees of Johnson &#x0026; Johnson. All authors are Observational Health Data Sciences and Informatics (OHDSI) collaborators.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ATC</term><def><p>Anatomical Therapeutic Chemical</p></def></def-item><def-item><term id="abb2">CDM</term><def><p>Common Data Model</p></def></def-item><def-item><term id="abb3">CPT4</term><def><p>Current Procedural Terminology, 4th Edition</p></def></def-item><def-item><term id="abb4">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb5">HCPCS</term><def><p>Healthcare Common Procedure Coding System</p></def></def-item><def-item><term id="abb6"><italic>ICD-10</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision</italic></p></def></def-item><def-item><term id="abb7"><italic>ICD-10-PCS</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision, Procedure Coding System</italic></p></def></def-item><def-item><term id="abb8"><italic>ICD-9</italic></term><def><p><italic>International Classification of Diseases, Ninth Revision</italic></p></def></def-item><def-item><term id="abb9"><italic>ICD10CM</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision, Clinical Modification</italic></p></def></def-item><def-item><term id="abb10"><italic>ICD10CN</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision, Chinese Version</italic></p></def></def-item><def-item><term id="abb11"><italic>ICD9Proc</italic></term><def><p><italic>International Classification of Diseases, Ninth Revision, Procedure Coding System</italic></p></def></def-item><def-item><term id="abb12"><italic>ICDO3</italic></term><def><p><italic>International Classification of Diseases for Oncology, Third Revision</italic></p></def></def-item><def-item><term id="abb13">LOINC</term><def><p>Logical Observation Identifiers Names and Codes</p></def></def-item><def-item><term id="abb14">NDC</term><def><p>National Drug Code</p></def></def-item><def-item><term id="abb15">OHDSI</term><def><p>Observational Health Data Sciences and Informatics</p></def></def-item><def-item><term id="abb16">OMOP</term><def><p>Observational Medical Outcomes Partnership</p></def></def-item><def-item><term id="abb17">SNOMED CT</term><def><p>Systematized Nomenclature of Medicine Clinical Terms</p></def></def-item><def-item><term id="abb18">SSSOM</term><def><p>Simple Standard for Sharing Ontological Mappings</p></def></def-item><def-item><term id="abb19">WHO</term><def><p>World Health Organization</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schuemie</surname><given-names>MJ</given-names> </name><name name-style="western"><surname>Madigan</surname><given-names>D</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>PB</given-names> </name><name name-style="western"><surname>Suchard</surname><given-names>MA</given-names> </name></person-group><article-title>Drawing reproducible conclusions from observational clinical data with OHDSI</article-title><source>Yearb Med Inform</source><year>2021</year><month>08</month><volume>30</volume><issue>1</issue><fpage>283</fpage><lpage>289</lpage><pub-id pub-id-type="doi">10.1055/s-0041-1726481</pub-id><pub-id pub-id-type="medline">33882595</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brown</surname><given-names>BB</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>C</given-names> </name><name name-style="western"><surname>McInnes</surname><given-names>E</given-names> </name><name name-style="western"><surname>Mays</surname><given-names>N</given-names> </name><name name-style="western"><surname>Young</surname><given-names>J</given-names> </name><name name-style="western"><surname>Haines</surname><given-names>M</given-names> </name></person-group><article-title>The effectiveness of clinical networks in improving quality of care and patient outcomes: a systematic review of quantitative and qualitative studies</article-title><source>BMC Health Serv Res</source><year>2016</year><month>08</month><day>8</day><volume>16</volume><issue>1</issue><fpage>360</fpage><pub-id pub-id-type="doi">10.1186/s12913-016-1615-z</pub-id><pub-id pub-id-type="medline">27613378</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>F</given-names> </name><name name-style="western"><surname>Panagiotakos</surname><given-names>D</given-names> </name></person-group><article-title>Real-world data: a brief review of the methods, applications, challenges and opportunities</article-title><source>BMC Med Res Methodol</source><year>2022</year><month>11</month><day>5</day><volume>22</volume><issue>1</issue><fpage>287</fpage><pub-id pub-id-type="doi">10.1186/s12874-022-01768-6</pub-id><pub-id pub-id-type="medline">36335315</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Schneeweiss</surname><given-names>S</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Bate</surname><given-names>A</given-names> </name><name name-style="western"><surname>Trifir&#x00F2;</surname><given-names>G</given-names> </name><name name-style="western"><surname>Bartels</surname><given-names>DB</given-names> </name></person-group><article-title>Choosing among common data models for real-world data analyses fit for making decisions about the effectiveness of medical products</article-title><source>Clin Pharmacol Ther</source><year>2020</year><month>04</month><volume>107</volume><issue>4</issue><fpage>827</fpage><lpage>833</lpage><pub-id pub-id-type="doi">10.1002/cpt.1577</pub-id><pub-id pub-id-type="medline">31330042</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Reich</surname><given-names>C</given-names> </name><name name-style="western"><surname>Ostropolets</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>P</given-names> </name><etal/></person-group><article-title>OHDSI Standardized Vocabularies-a large-scale centralized reference ontology for international data harmonization</article-title><source>J Am Med Inform Assoc</source><year>2024</year><month>02</month><day>16</day><volume>31</volume><issue>3</issue><fpage>583</fpage><lpage>590</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad247</pub-id><pub-id pub-id-type="medline">38175665</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name><name name-style="western"><surname>Duke</surname><given-names>JD</given-names> </name><name name-style="western"><surname>Shah</surname><given-names>NH</given-names> </name><etal/></person-group><article-title>Observational Health Data Sciences and Informatics (OHDSI): opportunities for observational researchers</article-title><source>Stud Health Technol Inform</source><year>2015</year><volume>216</volume><fpage>574</fpage><lpage>578</lpage><pub-id pub-id-type="medline">26262116</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liyanage</surname><given-names>H</given-names> </name><name name-style="western"><surname>Krause</surname><given-names>P</given-names> </name><name name-style="western"><surname>De Lusignan</surname><given-names>S</given-names> </name></person-group><article-title>Using ontologies to improve semantic interoperability in health data</article-title><source>J Innov Health Inform</source><year>2015</year><month>07</month><day>10</day><volume>22</volume><issue>2</issue><fpage>309</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.14236/jhi.v22i2.159</pub-id><pub-id pub-id-type="medline">26245245</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bodenreider</surname><given-names>O</given-names> </name><name name-style="western"><surname>Cornet</surname><given-names>R</given-names> </name><name name-style="western"><surname>Vreeman</surname><given-names>DJ</given-names> </name></person-group><article-title>Recent developments in clinical terminologies - SNOMED CT, LOINC, and RxNorm</article-title><source>Yearb Med Inform</source><year>2018</year><month>08</month><volume>27</volume><issue>1</issue><fpage>129</fpage><lpage>139</lpage><pub-id pub-id-type="doi">10.1055/s-0038-1667077</pub-id><pub-id pub-id-type="medline">30157516</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cui</surname><given-names>L</given-names> </name><name name-style="western"><surname>Bodenreider</surname><given-names>O</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>GQ</given-names> </name></person-group><article-title>Auditing SNOMED CT hierarchical relations based on lexical features of concepts in non-lattice subgraphs</article-title><source>J Biomed Inform</source><year>2018</year><month>02</month><volume>78</volume><fpage>177</fpage><lpage>184</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2017.12.010</pub-id><pub-id pub-id-type="medline">29274386</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>L</given-names> </name><name name-style="western"><surname>Abeysinghe</surname><given-names>R</given-names> </name></person-group><article-title>An automated approach for identifying erroneous IS-A relations in SNOMED CT</article-title><source>AMIA Jt Summits Transl Sci Proc</source><year>2024</year><volume>2024</volume><fpage>545</fpage><lpage>554</lpage><pub-id pub-id-type="medline">38827070</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rector</surname><given-names>AL</given-names> </name><name name-style="western"><surname>Brandt</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schneider</surname><given-names>T</given-names> </name></person-group><article-title>Getting the foot out of the pelvis: modeling problems affecting use of SNOMED CT hierarchies in practical applications</article-title><source>J Am Med Inform Assoc</source><year>2011</year><volume>18</volume><issue>4</issue><fpage>432</fpage><lpage>440</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2010-000045</pub-id><pub-id pub-id-type="medline">21515545</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kahn</surname><given-names>MG</given-names> </name><name name-style="western"><surname>Callahan</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Barnard</surname><given-names>J</given-names> </name><etal/></person-group><article-title>A harmonized data quality assessment terminology and framework for the secondary use of electronic health record data</article-title><source>EGEMS (Wash DC)</source><year>2016</year><volume>4</volume><issue>1</issue><fpage>1244</fpage><pub-id pub-id-type="doi">10.13063/2327-9214.1244</pub-id><pub-id pub-id-type="medline">27713905</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Ostropolets</surname><given-names>A</given-names> </name></person-group><article-title>ohdsi-studies/ConceptPrevalence</article-title><source>GitHub</source><access-date>2026-08-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/ohdsi-studies/ConceptPrevalence">https://github.com/ohdsi-studies/ConceptPrevalence</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="web"><source>Athena</source><access-date>2026-09-07</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://athena.ohdsi.org/search-terms/start">https://athena.ohdsi.org/search-terms/start</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Voss</surname><given-names>EA</given-names> </name><name name-style="western"><surname>Blacketer</surname><given-names>C</given-names> </name><name name-style="western"><surname>van Sandijk</surname><given-names>S</given-names> </name><etal/></person-group><article-title>European Health Data &#x0026; Evidence Network-learnings from building out a standardized international health data network</article-title><source>J Am Med Inform Assoc</source><year>2023</year><month>12</month><day>22</day><volume>31</volume><issue>1</issue><fpage>209</fpage><lpage>219</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocad214</pub-id><pub-id pub-id-type="medline">37952118</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="web"><article-title>Release planning</article-title><source>GitHub</source><access-date>2024-11-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/OHDSI/Vocabulary-v5.0/wiki/Release-planning">https://github.com/OHDSI/Vocabulary-v5.0/wiki/Release-planning</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="web"><article-title>Upcoming changes</article-title><source>GitHub</source><access-date>2024-11-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/OHDSI/Vocabulary-v5.0/wiki/Upcoming-changes">https://github.com/OHDSI/Vocabulary-v5.0/wiki/Upcoming-changes</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="web"><article-title>Releases &#x00B7; OHDSI/Vocabulary-v5.0</article-title><source>GitHub</source><access-date>2023-03-16</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/OHDSI/Vocabulary-v5.0/releases">https://github.com/OHDSI/Vocabulary-v5.0/releases</ext-link></comment></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>Dymshyts</surname><given-names>D</given-names> </name><name name-style="western"><surname>DeFalco</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ostropolets</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rao</surname><given-names>G</given-names> </name><name name-style="western"><surname>Shoaibi</surname><given-names>A</given-names> </name><name name-style="western"><surname>Blacketer</surname><given-names>C</given-names> </name></person-group><article-title>Evaluating the impact of different vocabulary versions on cohort definitions and CDM</article-title><year>2024</year><access-date>2026-08-30</access-date><publisher-name>Observational Health Data Sciences and Informatics</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.ohdsi.org/2024showcase-23/">https://www.ohdsi.org/2024showcase-23/</ext-link></comment></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="web"><article-title>Community contribution</article-title><source>GitHub</source><access-date>2024-11-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/OHDSI/Vocabulary-v5.0/wiki/Community-contribution">https://github.com/OHDSI/Vocabulary-v5.0/wiki/Community-contribution</ext-link></comment></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Matentzoglu</surname><given-names>N</given-names> </name><name name-style="western"><surname>Balhoff</surname><given-names>JP</given-names> </name><name name-style="western"><surname>Bello</surname><given-names>SM</given-names> </name><etal/></person-group><article-title>A Simple Standard for Sharing Ontological Mappings (SSSOM)</article-title><source>Database (Oxford)</source><year>2022</year><month>05</month><day>25</day><volume>2022</volume><fpage>baac035</fpage><pub-id pub-id-type="doi">10.1093/database/baac035</pub-id><pub-id pub-id-type="medline">35616100</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="web"><article-title>Release notes v20240830</article-title><source>GitHub</source><access-date>2025-06-12</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/OHDSI/Vocabulary-v5.0/releases/tag/v20240830_1725004358.000000">https://github.com/OHDSI/Vocabulary-v5.0/releases/tag/v20240830_1725004358.000000</ext-link></comment></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hur</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Feldman</surname><given-names>EL</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name></person-group><article-title>Ontology-based Brucella vaccine literature indexing and systematic analysis of gene-vaccine association network</article-title><source>BMC Immunol</source><year>2011</year><month>08</month><day>26</day><volume>12</volume><fpage>49</fpage><pub-id pub-id-type="doi">10.1186/1471-2172-12-49</pub-id><pub-id pub-id-type="medline">21871085</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ozg&#x00FC;r</surname><given-names>A</given-names> </name><name name-style="western"><surname>Xiang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Radev</surname><given-names>DR</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name></person-group><article-title>Mining of vaccine-associated IFN-&#x03B3; gene interaction networks using the Vaccine Ontology</article-title><source>J Biomed Semantics</source><year>2011</year><month>05</month><day>17</day><volume>2 Suppl 2</volume><issue>Suppl 2</issue><fpage>S8</fpage><pub-id pub-id-type="doi">10.1186/2041-1480-2-S2-S8</pub-id><pub-id pub-id-type="medline">21624163</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>C</given-names> </name><name name-style="western"><surname>He</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Kanjamala</surname><given-names>P</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>H</given-names> </name></person-group><article-title>Network-based analysis of vaccine-related associations reveals consistent knowledge with the Vaccine Ontology</article-title><source>J Biomed Semantics</source><year>2013</year><month>11</month><day>11</day><volume>4</volume><fpage>33</fpage><pub-id pub-id-type="doi">10.1186/2041-1480-4-33</pub-id><pub-id pub-id-type="medline">24209834</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>AY</given-names> </name><name name-style="western"><surname>Matheny</surname><given-names>M</given-names> </name><name name-style="western"><surname>Williams</surname><given-names>A</given-names> </name><name name-style="western"><surname>You</surname><given-names>SC</given-names> </name></person-group><article-title>Medical device standard terminology overview, comparison and analysis</article-title><source>Observational Health Data Sciences and Informatics</source><year>2024</year><access-date>2026-08-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ohdsi.org/2024showcase-30/">https://www.ohdsi.org/2024showcase-30/</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amadi</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kiwuwa-Muyingo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhattacharjee</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Making metadata machine-readable as the first step to providing findable, accessible, interoperable, and reusable population health data: framework development and implementation study</article-title><source>Online J Public Health Inform</source><year>2024</year><month>08</month><day>1</day><volume>16</volume><fpage>e56237</fpage><pub-id pub-id-type="doi">10.2196/56237</pub-id><pub-id pub-id-type="medline">39088253</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhattacharjee</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kiwuwa-Muyingo</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kanjala</surname><given-names>C</given-names> </name><etal/></person-group><article-title>INSPIRE datahub: a pan-African integrated suite of services for harmonising longitudinal population health data using OHDSI tools</article-title><source>Front Digit Health</source><year>2024</year><volume>6</volume><fpage>1329630</fpage><pub-id pub-id-type="doi">10.3389/fdgth.2024.1329630</pub-id><pub-id pub-id-type="medline">38347885</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="web"><article-title>Introducing the LOINC Ontology: a LOINC and SNOMED CT interoperability solution</article-title><source>SNOMED International</source><access-date>2024-11-11</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://loincsnomed.org/">https://loincsnomed.org/</ext-link></comment></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carroll</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>WK</given-names> </name><name name-style="western"><surname>Eyler</surname><given-names>AE</given-names> </name><etal/></person-group><article-title>Portability of an algorithm to identify rheumatoid arthritis in electronic health records</article-title><source>J Am Med Inform Assoc</source><year>2012</year><month>06</month><volume>19</volume><issue>e1</issue><fpage>e162</fpage><lpage>e169</lpage><pub-id pub-id-type="doi">10.1136/amiajnl-2011-000583</pub-id><pub-id pub-id-type="medline">22374935</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pacheco</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Rasmussen</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Kiefer</surname><given-names>RC</given-names> </name><etal/></person-group><article-title>A case study evaluating the portability of an executable computable phenotype algorithm across multiple institutions and electronic health record environments</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>11</month><day>1</day><volume>25</volume><issue>11</issue><fpage>1540</fpage><lpage>1546</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocy101</pub-id><pub-id pub-id-type="medline">30124903</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sohn</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wi</surname><given-names>CI</given-names> </name><etal/></person-group><article-title>Clinical documentation variations and NLP system portability: a case study in asthma birth cohorts across institutions</article-title><source>J Am Med Inform Assoc</source><year>2018</year><month>03</month><day>1</day><volume>25</volume><issue>3</issue><fpage>353</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocx138</pub-id><pub-id pub-id-type="medline">29202185</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Ostropolets</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>R</given-names> </name><name name-style="western"><surname>Spotnitz</surname><given-names>M</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>R</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name></person-group><article-title>Phenotype algorithm and data source reporting in top clinical journals: where we are and where should we go</article-title><source>Observational Health Data Sciences and Informatics</source><year>2020</year><access-date>2026-05-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ohdsi.org/2020-global-symposium-showcase-25/">https://www.ohdsi.org/2020-global-symposium-showcase-25/</ext-link></comment></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhargava</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>T</given-names> </name><name name-style="western"><surname>Quine</surname><given-names>DB</given-names> </name><name name-style="western"><surname>Hauser</surname><given-names>RG</given-names> </name></person-group><article-title>A 20-year evaluation of LOINC in the United States&#x2019; largest integrated health system</article-title><source>Arch Pathol Lab Med</source><year>2020</year><month>04</month><volume>144</volume><issue>4</issue><fpage>478</fpage><lpage>484</lpage><pub-id pub-id-type="doi">10.5858/arpa.2019-0055-OA</pub-id><pub-id pub-id-type="medline">31469586</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ostropolets</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ryan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hripcsak</surname><given-names>G</given-names> </name></person-group><article-title>Phenotyping in distributed data networks: selecting the right codes for the right patients</article-title><source>AMIA Annu Symp Proc</source><year>2023</year><volume>2022</volume><fpage>826</fpage><lpage>835</lpage><pub-id pub-id-type="medline">37128407</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rasmussen</surname><given-names>LV</given-names> </name><name name-style="western"><surname>Brandt</surname><given-names>PS</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Considerations for improving the portability of electronic health record-based phenotype algorithms</article-title><source>AMIA Annu Symp Proc</source><year>2020</year><volume>2019</volume><fpage>755</fpage><lpage>764</lpage><pub-id pub-id-type="medline">32308871</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dalton</surname><given-names>MK</given-names> </name><name name-style="western"><surname>Sokas</surname><given-names>CM</given-names> </name><name name-style="western"><surname>Castillo-Angeles</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Defining the emergency general surgery patient population in the era of ICD-10: evaluating an established crosswalk from ICD-9 to ICD-10 diagnosis codes</article-title><source>J Trauma Acute Care Surg</source><year>2023</year><month>12</month><day>1</day><volume>95</volume><issue>6</issue><fpage>899</fpage><lpage>904</lpage><pub-id pub-id-type="doi">10.1097/TA.0000000000004050</pub-id><pub-id pub-id-type="medline">37381148</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Philipp</surname><given-names>P</given-names> </name><name name-style="western"><surname>Veloso</surname><given-names>J</given-names> </name><name name-style="western"><surname>Appenzeller</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hartz</surname><given-names>T</given-names> </name><name name-style="western"><surname>Beyerer</surname><given-names>J</given-names> </name></person-group><article-title>Evaluation of an automated mapping from ICD-10 to SNOMED CT</article-title><source>2022 International Conference on Computational Science and Computational Intelligence (CSCI)</source><year>2022</year><publisher-name>IEEE</publisher-name><pub-id pub-id-type="doi">10.1109/CSCI58124.2022.00287</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Rodrigues</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Robinson</surname><given-names>D</given-names> </name><name name-style="western"><surname>Della Mea</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Semantic alignment between ICD-11 and SNOMED CT</article-title><source>Studies in Health Technology and Informatics</source><year>2015</year><publisher-name>IOS Press</publisher-name><fpage>790</fpage><lpage>794</lpage><pub-id pub-id-type="doi">10.3233/978-1-61499-564-7-790</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zheng</surname><given-names>L</given-names> </name><name name-style="western"><surname>He</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>D</given-names> </name><etal/></person-group><article-title>A review of auditing techniques for the Unified Medical Language System</article-title><source>J Am Med Inform Assoc</source><year>2020</year><month>10</month><day>1</day><volume>27</volume><issue>10</issue><fpage>1625</fpage><lpage>1638</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaa108</pub-id><pub-id pub-id-type="medline">32766692</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tsaneva</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sabou</surname><given-names>M</given-names> </name></person-group><article-title>Enhancing human-in-the-loop ontology curation results through task design</article-title><source>ACM J Data Inf Qual</source><year>2024</year><month>03</month><volume>16</volume><issue>1</issue><fpage>1</fpage><lpage>25</lpage><pub-id pub-id-type="doi">10.1145/3626960</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>DH</given-names> </name><name name-style="western"><surname>Lau</surname><given-names>FY</given-names> </name></person-group><article-title>An exploratory analysis of SNOMED CT national editions</article-title><source>J Am Med Inform Assoc</source><year>2026</year><month>02</month><volume>33</volume><issue>2</issue><fpage>383</fpage><lpage>393</lpage><pub-id pub-id-type="doi">10.1093/jamia/ocaf184</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Resnick</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Hitt</surname><given-names>J</given-names> </name><name name-style="western"><surname>McCray</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Semantic relations: extending SNOMED CT and Solor</article-title><source>Appl Clin Inform</source><year>2025</year><month>08</month><volume>16</volume><issue>4</issue><fpage>1263</fpage><lpage>1270</lpage><pub-id pub-id-type="doi">10.1055/a-2606-9411</pub-id><pub-id pub-id-type="medline">41043478</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Adams</surname><given-names>MC</given-names> </name><name name-style="western"><surname>Perkins</surname><given-names>ML</given-names> </name><name name-style="western"><surname>Hudson</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Breaking digital health barriers through a large language model-based tool for automated observational medical outcomes partnership mapping: development and validation study</article-title><source>J Med Internet Res</source><year>2025</year><month>05</month><day>15</day><volume>27</volume><fpage>e69004</fpage><pub-id pub-id-type="doi">10.2196/69004</pub-id><pub-id pub-id-type="medline">40146872</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Barros Vanzin</surname><given-names>VJ</given-names> </name><name name-style="western"><surname>de Abreu Moreira</surname><given-names>D</given-names> </name><name name-style="western"><surname>Marcondes Marcacini</surname><given-names>R</given-names> </name></person-group><article-title>LLM-based approaches for automated vocabulary mapping between SIGTAP and OMOP CDM concepts</article-title><source>Artif Intell Med</source><year>2025</year><month>10</month><volume>168</volume><fpage>103204</fpage><pub-id pub-id-type="doi">10.1016/j.artmed.2025.103204</pub-id><pub-id pub-id-type="medline">40706107</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Katsch</surname><given-names>F</given-names> </name><name name-style="western"><surname>M&#x00E9;sz&#x00E1;ros</surname><given-names>&#x00C1;</given-names> </name><name name-style="western"><surname>H&#x00E9;ja</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hussein</surname><given-names>R</given-names> </name><name name-style="western"><surname>Duftschmid</surname><given-names>G</given-names> </name></person-group><article-title>Semiautomatic mapping of a national drug terminology to standardised OMOP drug concepts using publicly available supplementary information</article-title><source>BMC Med Res Methodol</source><year>2025</year><month>09</month><day>26</day><volume>25</volume><issue>1</issue><fpage>213</fpage><pub-id pub-id-type="doi">10.1186/s12874-025-02669-0</pub-id><pub-id pub-id-type="medline">41013235</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kimura</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kawakami</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Inoue</surname><given-names>S</given-names> </name><name name-style="western"><surname>Okajima</surname><given-names>A</given-names> </name></person-group><article-title>Mapping drug terms via integration of a retrieval-augmented generation algorithm with a large language model</article-title><source>Healthc Inform Res</source><year>2024</year><month>10</month><volume>30</volume><issue>4</issue><fpage>355</fpage><lpage>363</lpage><pub-id pub-id-type="doi">10.4258/hir.2024.30.4.355</pub-id><pub-id pub-id-type="medline">39551922</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kimura</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kawakami</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Inoue</surname><given-names>S</given-names> </name><name name-style="western"><surname>Okajima</surname><given-names>A</given-names> </name></person-group><article-title>A dataset for mapping the Japanese drugs to RxNorm standard concepts</article-title><source>Data Brief</source><year>2025</year><volume>59</volume><fpage>111418</fpage><pub-id pub-id-type="doi">10.1016/j.dib.2025.111418</pub-id><pub-id pub-id-type="medline">40124300</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="web"><article-title>Standardized vocabularies</article-title><source>Observational Health Data Sciences and Informatics (OHDSI)</source><access-date>2026-09-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://athena.ohdsi.org/vocabulary/list">https://athena.ohdsi.org/vocabulary/list</ext-link></comment></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="web"><article-title>Network prevalence count</article-title><source>ATLAS</source><access-date>2026-09-02</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://atlas-demo.ohdsi.org/#/datasources">https://atlas-demo.ohdsi.org/#/datasources</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Observational Health Data Sciences and Informatics (OHDSI) Vocabularies&#x2019; landscape assessment survey questions and use of terminologies within the OHDSI Standardized Vocabularies in the data.</p><media xlink:href="medinform_v14i1e92727_app1.docx" xlink:title="DOCX File, 31 KB"/></supplementary-material></app-group></back></article>