<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "http://dtd.nlm.nih.gov/publishing/2.0/journalpublishing.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="2.0">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">JMI</journal-id>
      <journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id>
      <journal-title>JMIR Medical Informatics</journal-title>
      <issn pub-type="epub">2291-9694</issn>
      <publisher>
        <publisher-name>JMIR Publications</publisher-name>
        <publisher-loc>Toronto, Canada</publisher-loc>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="publisher-id">v14i1e82924</article-id>
      <article-id pub-id-type="pmid">42470189</article-id>
      <article-id pub-id-type="doi">10.2196/82924</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Original Paper</subject>
        </subj-group>
        <subj-group subj-group-type="article-type">
          <subject>Original Paper</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Benchmarking Fast Healthcare Interoperability Resources–Based Analytics: Quantitative Study of RESTful Server Queries and Big Data Engines</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Benis</surname>
            <given-names>Arriel</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Wiedekopf</surname>
            <given-names>Joshua</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Thawillarp</surname>
            <given-names>Supharerk</given-names>
          </name>
        </contrib>
        <contrib contrib-type="reviewer">
          <name>
            <surname>Al-Agil</surname>
            <given-names>Mohammad</given-names>
          </name>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib id="contrib1" contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Gulden</surname>
            <given-names>Christian</given-names>
          </name>
          <degrees>Dr</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <address>
            <institution>Lehrstuhl für Medizinische Informatik</institution>
            <institution>Institut für Medizininformatik, Biometrie und Epidemiologie</institution>
            <institution>Friedrich-Alexander-Universität-Erlangen-Nürnberg</institution>
            <addr-line>Wetterkreuz 15</addr-line>
            <addr-line>Erlangen</addr-line>
            <country>Germany</country>
            <phone>49 9131 85 677</phone>
            <email>christian.gulden@fau.de</email>
          </address>
          <xref rid="aff2" ref-type="aff">2</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-1261-3691</ext-link>
        </contrib>
        <contrib id="contrib2" contrib-type="author">
          <name name-style="western">
            <surname>Kampf</surname>
            <given-names>Marvin</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-9108-0469</ext-link>
        </contrib>
        <contrib id="contrib3" contrib-type="author">
          <name name-style="western">
            <surname>Kraska</surname>
            <given-names>Detlef</given-names>
          </name>
          <degrees>Dr</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-2174-2532</ext-link>
        </contrib>
        <contrib id="contrib4" contrib-type="author">
          <name name-style="western">
            <surname>Grimes</surname>
            <given-names>John</given-names>
          </name>
          <degrees>BIT</degrees>
          <xref rid="aff4" ref-type="aff">4</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0002-9575-7641</ext-link>
        </contrib>
        <contrib id="contrib5" contrib-type="author">
          <name name-style="western">
            <surname>Ganslandt</surname>
            <given-names>Thomas</given-names>
          </name>
          <degrees>Prof Dr</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6864-8936</ext-link>
        </contrib>
        <contrib id="contrib6" contrib-type="author">
          <name name-style="western">
            <surname>Prokosch</surname>
            <given-names>Hans-Ulrich</given-names>
          </name>
          <degrees>Prof Dr</degrees>
          <xref rid="aff1" ref-type="aff">1</xref>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0001-6200-753X</ext-link>
        </contrib>
        <contrib id="contrib7" contrib-type="author">
          <name name-style="western">
            <surname>Seuchter</surname>
            <given-names>Susanne A.</given-names>
          </name>
          <degrees>BSc</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0006-2890-0280</ext-link>
        </contrib>
        <contrib id="contrib8" contrib-type="author">
          <name name-style="western">
            <surname>Mang</surname>
            <given-names>Jonathan M.</given-names>
          </name>
          <degrees>Dr</degrees>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-0518-4710</ext-link>
        </contrib>
        <contrib id="contrib9" contrib-type="author">
          <name name-style="western">
            <surname>Pallaoro</surname>
            <given-names>Peter</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff5" ref-type="aff">5</xref>
          <xref rid="aff6" ref-type="aff">6</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0000-0003-4808-0700</ext-link>
        </contrib>
        <contrib id="contrib10" contrib-type="author">
          <name name-style="western">
            <surname>Volkmer</surname>
            <given-names>Paul-Christian</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff7" ref-type="aff">7</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0007-0967-9696</ext-link>
        </contrib>
        <contrib id="contrib11" contrib-type="author">
          <name name-style="western">
            <surname>Ziegler</surname>
            <given-names>Jasmin</given-names>
          </name>
          <degrees>MSc</degrees>
          <xref rid="aff2" ref-type="aff">2</xref>
          <xref rid="aff3" ref-type="aff">3</xref>
          <ext-link ext-link-type="orcid">https://orcid.org/0009-0005-5362-5228</ext-link>
        </contrib>
      </contrib-group>
      <aff id="aff1">
        <label>1</label>
        <institution>Lehrstuhl für Medizinische Informatik</institution>
        <institution>Institut für Medizininformatik, Biometrie und Epidemiologie</institution>
        <institution>Friedrich-Alexander-Universität-Erlangen-Nürnberg</institution>
        <addr-line>Erlangen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff2">
        <label>2</label>
        <institution>Bavarian Cancer Research Center (BZKF)</institution>
        <addr-line>Erlangen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff3">
        <label>3</label>
        <institution>Medical Center for Information and Communication Technology</institution>
        <institution>Universitätsklinikum Erlangen</institution>
        <addr-line>Erlangen</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff4">
        <label>4</label>
        <institution>CSIRO Health and Biosecurity</institution>
        <institution>Australian e-Health Research Centre</institution>
        <addr-line>Brisbane</addr-line>
        <country>Australia</country>
      </aff>
      <aff id="aff5">
        <label>5</label>
        <institution>Chair of Medical Informatics, Institute for AI and Informatics in Medicine (AIIM), TUM University Hospital</institution>
        <institution>TUM School of Medicine and Health</institution>
        <institution>Technical University of Munich</institution>
        <addr-line>Munich</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff6">
        <label>6</label>
        <institution>Data Integration Center, TUM University Hospital</institution>
        <institution>TUM School of Medicine and Health</institution>
        <institution>Technical University of Munich</institution>
        <addr-line>Munich</addr-line>
        <country>Germany</country>
      </aff>
      <aff id="aff7">
        <label>7</label>
        <institution>Anneliese Pohl Krebszentrum Marburg</institution>
        <institution>Comprehensive Cancer Center</institution>
        <institution>Universitätsklinikum Gießen und Marburg</institution>
        <addr-line>Marburg</addr-line>
        <country>Germany</country>
      </aff>
      <author-notes>
        <corresp>Corresponding Author: Christian Gulden <email>christian.gulden@fau.de</email></corresp>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>17</day>
        <month>7</month>
        <year>2026</year>
      </pub-date>
      <volume>14</volume>
      <elocation-id>e82924</elocation-id>
      <history>
        <date date-type="received">
          <day>24</day>
          <month>8</month>
          <year>2025</year>
        </date>
        <date date-type="rev-request">
          <day>11</day>
          <month>11</month>
          <year>2025</year>
        </date>
        <date date-type="rev-recd">
          <day>10</day>
          <month>5</month>
          <year>2026</year>
        </date>
        <date date-type="accepted">
          <day>31</day>
          <month>5</month>
          <year>2026</year>
        </date>
      </history>
      <copyright-statement>©Christian Gulden, Marvin Kampf, Detlef Kraska, John Grimes, Thomas Ganslandt, Hans-Ulrich Prokosch, Susanne A. Seuchter, Jonathan M. Mang, Peter Pallaoro, Paul-Christian Volkmer, Jasmin Ziegler. Originally published in JMIR Medical Informatics (https://medinform.jmir.org), 17.07.2026.</copyright-statement>
      <copyright-year>2026</copyright-year>
      <license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/">
        <p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (https://creativecommons.org/licenses/by/4.0/), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on https://medinform.jmir.org/, as well as this copyright and license information must be included.</p>
      </license>
      <self-uri xlink:href="https://medinform.jmir.org/2026/1/e82924" xlink:type="simple"/>
      <abstract>
        <sec sec-type="background">
          <title>Background</title>
          <p>Electronic health records offer vast clinical data for health care research, but interoperability challenges often hinder comprehensive analysis. The Health Level Seven Fast Healthcare Interoperability Resources (FHIR) standard addresses these challenges, although its nested and interconnected resource format can be complex for analytics. Several tools have emerged to facilitate analytical access, either by querying FHIR servers via representational state transfer (REST) APIs or encoding resources in relational formats. However, the performance implications of these methods remain largely unexplored.</p>
        </sec>
        <sec sec-type="objective">
          <title>Objective</title>
          <p>This study aimed to benchmark the performance characteristics of different FHIR-based analytical approaches comparing REST API queries against SQL- and Spark-based big data frameworks operating on FHIR-encoded data.</p>
        </sec>
        <sec sec-type="methods">
          <title>Methods</title>
          <p>We benchmarked the FHIR-PYrate library, which interfaces with a FHIR server’s REST API, against Pathling, a library built for analytics based on Apache Spark, and Trino, a general-purpose SQL query engine. We defined and implemented multiple queries in each engine using 3 common analytics scenarios—data aggregation, counting, and extraction. Execution times were measured across Synthea-generated datasets of increasing size.</p>
        </sec>
        <sec sec-type="results">
          <title>Results</title>
          <p>On the largest dataset, containing 71,285,064 FHIR resources, Trino completed the aggregate query more than 12,000 times faster, and Pathling did so approximately 500 times faster than FHIR-PYrate. On average across all queries, Trino outperformed FHIR-PYrate, executing extraction queries 33 times faster and count queries 1.8 times faster. Pathling achieved a 2.6-time speedup for extraction queries, but FHIR-PYrate was approximately 13 times faster for count queries.</p>
        </sec>
        <sec sec-type="conclusions">
          <title>Conclusions</title>
          <p>While the REST-based FHIR search API is useful for standard queries and retrieving specific patient records and can outperform alternatives for some count queries, it generally lacks the performance and expressiveness needed for complex analytics. In contrast, alternative engines such as Trino and Pathling demonstrated substantial performance advantages for these scenarios.</p>
        </sec>
      </abstract>
      <kwd-group>
        <kwd>Fast Healthcare Interoperability Resources</kwd>
        <kwd>FHIR</kwd>
        <kwd>benchmark</kwd>
        <kwd>performance</kwd>
        <kwd>big data</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec sec-type="introduction">
      <title>Introduction</title>
      <p>Electronic health records provide digital means to store and retrieve clinical data for millions of patients. Leveraging these real-world data provides an exciting opportunity for health care research [<xref ref-type="bibr" rid="ref1">1</xref>]. However, the heterogeneity of electronic health record systems often results in interoperability challenges, hindering comprehensive analysis—especially across institutional boundaries [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. This is where the Health Level Seven Fast Healthcare Interoperability Resources (FHIR) can help by providing a standardized framework for modeling and exchanging health care data [<xref ref-type="bibr" rid="ref4">4</xref>]. The standard describes a data model consisting of resources that represent concepts in the health care domain. These resources are connected via references and form a graph. For example, a FHIR “Condition” resource representing a medical diagnosis may reference a “Patient” resource to which it applies and a “Practitioner” resource identifying the person recording the diagnosis. The FHIR standard also defines an HTTP representational state transfer (REST) interface and its semantics. This interface is implemented by FHIR servers to support the creation, reading, updating, and deletion of resources. Additionally, FHIR Search provides a comprehensive framework for filtering and retrieving relevant resources [<xref ref-type="bibr" rid="ref5">5</xref>].</p>
      <p>The increased adoption of FHIR is driving its use for analytic use cases, including decision support, feasibility research, and epidemiological research [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. When conducting analytical research on FHIR data, familiarity with the graphlike structure, conventions, data types, and search semantics is required. From a purely syntactic perspective, data science and machine learning are usually conducted on tabular data, whereas FHIR resources are nested and interconnected—often encoded as JSON or XML file formats. Due to this gap, several projects aim to convert FHIR resources to a structure more suitable for analytical use cases. These approaches either interact with FHIR servers in a combination of FHIR Search and local postprocessing [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref11">11</xref>] or encode and store resources in a relational or columnar format first [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref15">15</xref>]. Nevertheless, the performance implications of these methods remain largely unexplored.</p>
      <p>In this work, we compared analytical approaches that operate on Health Level Seven FHIR as the underlying data representation but differ in how the data are accessed and processed, ranging from REST-based querying of FHIR servers to SQL- and Spark-based querying of tabular-encoded FHIR datasets.</p>
    </sec>
    <sec sec-type="methods">
      <title>Methods</title>
      <sec>
        <title>Engines and Queries</title>
        <p>We compared the performance of 3 different approaches for conducting analytics on persisted FHIR resources:</p>
        <list list-type="bullet">
          <list-item>
            <p>FHIR-PYrate (version 0.2.3) [<xref ref-type="bibr" rid="ref9">9</xref>]: a Python library that interacts with a FHIR server’s REST API to filter and download resources and with FHIRPath [<xref ref-type="bibr" rid="ref16">16</xref>] to extract individual attributes</p>
          </list-item>
          <list-item>
            <p>Pathling (version 7.2.0) [<xref ref-type="bibr" rid="ref12">12</xref>]: a Python, Scala, Java, and R library used to query FHIR data previously encoded as Apache Spark [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref18">18</xref>] datasets; like FHIR-PYrate, it allows for the use of FHIRPath to describe the tabular data to be extracted from the resources</p>
          </list-item>
          <list-item>
            <p>Trino (version 478) [<xref ref-type="bibr" rid="ref19">19</xref>]: a distributed American National Standards Institute (ANSI) SQL-compliant federated query engine</p>
          </list-item>
        </list>
        <p>The above-mentioned engines were chosen to allow for comparison between accessing resources directly via a FHIR server’s search features, via a more FHIR-native way using FHIRPath expressions, and via SQL—often considered the lingua franca of data analytics [<xref ref-type="bibr" rid="ref20">20</xref>]. Comparing Pathling and Trino additionally allowed us to benchmark 2 frameworks based on the same Delta Lake table format [<xref ref-type="bibr" rid="ref21">21</xref>] but using different query engines. This storage-compute decoupling approach is increasingly adopted in modern data warehouse designs [<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref23">23</xref>].</p>
        <p>We compared the performance of each framework in 3 common analytics scenarios:</p>
        <list list-type="bullet">
          <list-item>
            <p>Aggregation: summarizing data to provide statistics such as mean, median, minimum, maximum, sum, and count of values</p>
          </list-item>
          <list-item>
            <p>Counting: counting the number of resources that satisfy given criteria (eg, this is a common use in feasibility studies to assess the number of potentially eligible patients)</p>
          </list-item>
          <list-item>
            <p>Extraction: retrieving a subset of relevant data in a tabular format filtered to answer a research question</p>
          </list-item>
        </list>
        <p>For the aggregation scenario, we counted the number of FHIR “Observation” resources grouped by their code (eg, the total number of hemoglobin level observations across all patients). This is a common query for inventorying and presenting total available data, such as in a dashboard. As FHIR Search lacks built-in aggregation capabilities, the implementation using FHIR-PYrate first had to download all “Observation” resources into an in-memory pandas DataFrame and execute the aggregation using the pandas “groupby” function. Therefore, this query represents a fundamentally challenging scenario for standard-based FHIR Search interfaces as the FHIR Search specification does not define native server-side aggregation operations. The performance characteristics are determined by the need for client-side retrieval and processing of all available resources.</p>
        <p>For both counting and extracting, we implemented 3 queries to retrieve data. The queries were chosen to be close to real-world use cases while at the same time being compatible with the synthetic data against which they were executed. Further, the queries were increasingly complex, integrating multiple resource types and adding multiple filter conditions. The queries were also created in a way that would allow them to be represented as a single FHIR Search request. This made comparison between the engines easier as no additional postprocessing had to take place inside the benchmarking code.</p>
        <p>The descriptions of the 3 queries are provided in <xref ref-type="table" rid="table1">Table 1</xref>. For the count scenario, only the number of patients satisfying the filter criteria was returned, whereas for the extraction and aggregation scenarios, a projection of resource elements in tabular form was written to a CSV file.</p>
        <table-wrap position="float" id="table1">
          <label>Table 1</label>
          <caption>
            <p>Description of the 3 queries used to benchmark the engines for the count and extraction scenarios.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="160"/>
            <col width="840"/>
            <thead>
              <tr valign="top">
                <td>Query name</td>
                <td>Description</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Gender-age</td>
                <td>All patients whose gender was set to “female” and whose birth date was on or after January 1, 1970</td>
              </tr>
              <tr valign="top">
                <td>Diabetes</td>
                <td>All patients born after January 1, 1970, with a diagnosis of diabetes documented for an encounter taking place on or after January 1, 2020</td>
              </tr>
              <tr valign="top">
                <td>Hemoglobin</td>
                <td>All patients with a hemoglobin laboratory value given either in mass per volume format that was higher than 25 g/dL or as a relative percentage of HbA<sub>1c</sub><sup>a</sup> compared to total blood hemoglobin that was higher than 5%</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table1fn1">
              <p><sup>a</sup>HbA<sub>1c</sub>: hemoglobin A<sub>1c</sub>.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>The main difference between the count queries and the aggregation and extraction ones is that the former only returns a single scalar result value to the caller, so the impact of large data transfers between the benchmark client and the query implementation is significantly reduced.</p>
        <p>To compare the scalability of the different engines across all scenarios, we executed the aggregate, count, and extract queries on progressively larger FHIR datasets generated using Synthea [<xref ref-type="bibr" rid="ref24">24</xref>]. The Synthea command-line tool was used to generate a reproducible dataset of 1000, 5000, 10,000, 50,000, and 100,000 records. These resources were stored as JSON-encoded files in the local file system. For each record count, the scripted benchmarking workflow was as follows:</p>
        <list list-type="bullet">
          <list-item>
            <p>Start all the required software components.</p>
          </list-item>
          <list-item>
            <p>Encode the FHIR resources as Delta Lake tables using the Pathling server’s bulk import operation.</p>
          </list-item>
          <list-item>
            <p>Run Delta Lake–specific OPTIMIZE and VACUUM commands to optimize the file layout.</p>
          </list-item>
          <list-item>
            <p>Load the synthetic data into the Blaze FHIR server followed by the HAPI FHIR server.</p>
          </list-item>
          <list-item>
            <p>Run PostgreSQL’s VACUUM FULL and ANALYZE commands against the largest HAPI FHIR PostgreSQL tables.</p>
          </list-item>
          <list-item>
            <p>Stop the Pathling server as it is only used for encoding and persisting the resources in MinIO object storage as tabular data.</p>
          </list-item>
          <list-item>
            <p>For each query type (count, extract, and aggregate), repeat the following process 10 times: (1) record the total time taken per engine and query from the moment the query is submitted until the CSV file is fully written to disk; (2) execute the SQL queries against Trino consecutively, download all results, and save them as a CSV file; (3) restart the Trino container and MinIO and wait 30 seconds for start-up to complete to ensure that the system stabilizes and background tasks are complete before measurements begin; (4) execute the queries using Pathling and save them as a CSV file; (5) reset the cache, force garbage collection, reinitialize Spark, and wait 30 seconds; (6) execute the queries using FHIR-PYrate, saving the results as CSV files; and (7) restart the FHIR server container and wait 30 seconds.</p>
          </list-item>
        </list>
        <p>Restarting the FHIR server and Trino clears in-memory caches and frees up unused memory for the operating system. Each benchmark run started with a different query in a round-robin fashion, ensuring that no query consistently preceded others and minimizing potential order effects. The workflow was implemented using Python (version 3.12; Python Software Foundation), and runtimes were measured using a monotonic performance counter. The benchmarks were executed on a single virtual machine (type CCX53) hosted on Hetzner Cloud running Ubuntu (version 24.04; Canonical Ltd). The machine was equipped with 32 dedicated AMD EPYC (Milan) virtual central processing units (CPUs), 128 GB of RAM, and a 600-GB Non-Volatile Memory Express solid-state drive.</p>
        <p><xref rid="figure1" ref-type="fig">Figure 1</xref> shows the application setup on the virtual machine. For the Trino benchmark, a Hive metastore (The Apache Software Foundation) was used to register the resource tables and their location in the MinIO object storage. However, the metastore was not part of the critical path for the benchmark and was only used by Trino to determine the physical location of the Delta Lake tables.</p>
        <fig id="figure1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Application configuration of the benchmarking virtual machine. Both the Hive metastore and the Hive metastore PostgreSQL DB were not on the critical benchmark data path and are drawn in lighter gray. FHIR: Fast Healthcare Interoperability Resources.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig1.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>The individual application container images and versions are listed in <xref ref-type="table" rid="table2">Table 2</xref>. The table also includes the memory limit set for each application: for Trino, Blaze, and the HAPI FHIR JPA server, the memory limits were set via the “-Xmx64g” Java Virtual Machine option. For the remaining applications, the memory limits were set using group memory limits on the containers [<xref ref-type="bibr" rid="ref25">25</xref>].</p>
        <table-wrap position="float" id="table2">
          <label>Table 2</label>
          <caption>
            <p>Applications and their respective container image, version, and memory limit used for the benchmark.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="260"/>
            <col width="300"/>
            <col width="300"/>
            <col width="140"/>
            <thead>
              <tr valign="top">
                <td>Application</td>
                <td>Container image</td>
                <td>Version</td>
                <td>Memory limit</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>Trino</td>
                <td>docker.io/trinodb/trino</td>
                <td>478</td>
                <td>64 GB</td>
              </tr>
              <tr valign="top">
                <td>Blaze</td>
                <td>docker.io/samply/blaze</td>
                <td>1.2.0</td>
                <td>64 GB</td>
              </tr>
              <tr valign="top">
                <td>Pathling server</td>
                <td>docker.io/aehrc/pathling</td>
                <td>7.2.0</td>
                <td>56 GB</td>
              </tr>
              <tr valign="top">
                <td>MinIO</td>
                <td>docker.io/minio/minio</td>
                <td>RELEASE.2025-09-07T16-13-09Z</td>
                <td>8 GB</td>
              </tr>
              <tr valign="top">
                <td>Hive metastore</td>
                <td>docker.io/apache/hive</td>
                <td>4.0.0</td>
                <td>2 GB</td>
              </tr>
              <tr valign="top">
                <td>Hive metastore Database</td>
                <td>docker.io/library/postgres</td>
                <td>18.1</td>
                <td>1 GB</td>
              </tr>
              <tr valign="top">
                <td>HAPI FHIR<sup>a</sup> server</td>
                <td>docker.io/hapiproject/hapi</td>
                <td>8.6.0-1</td>
                <td>48 GB</td>
              </tr>
              <tr valign="top">
                <td>HAPI FHIR server Database</td>
                <td>docker.io/library/postgres</td>
                <td>18.1</td>
                <td>16 GB</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table2fn1">
              <p><sup>a</sup>FHIR: Fast Healthcare Interoperability Resources.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>The Blaze server was tuned following the production configuration guide [<xref ref-type="bibr" rid="ref26">26</xref>], and the HAPI FHIR server PostgreSQL database was configured using the PGTune online tool [<xref ref-type="bibr" rid="ref27">27</xref>] for data warehouse workloads. For Pathling, we deviated from the default configuration by setting the Parquet compression codec to Zstandard level 9 and the serializer to KryoSerializer. The complete benchmark configuration is available in the public source code repository.</p>
      </sec>
      <sec>
        <title>Measuring Resource Use for Data Loading</title>
        <p>As a first step, the FHIR resources generated by the Synthea command-line interface had to be loaded into the FHIR servers (for HAPI and Blaze) or encoded as Delta Lake tables (for Pathling and Trino). To load the resources into the servers, we used blazectl (version 1.2.0) with client concurrency set to 32; for Delta Lake, we used the Pathling server’s FHIR bulk import API. The former tool reports the import duration after completion; for the latter, we used the Unix time binary to measure the duration of the synchronous import triggered via curl.</p>
        <p>We used cAdvisor (version 0.53.0) to record the container’s CPU and memory use during the data loading. The metrics were stored in a Prometheus time-series database (version 3.7.3), retrieved when the imports finished, and stored in a CSV file for later visualization. Prometheus was configured to scrape cAdvisor metrics every 5 seconds.</p>
        <p>Per-container disk use was measured after the import was completed using the Docker command-line interface.</p>
      </sec>
      <sec>
        <title>Comparing FHIR Server Performance</title>
        <p>A separate benchmark was conducted to compare the performance of the 2 FHIR server implementations: Blaze and the HAPI FHIR server. The original hemoglobin query, which included multiple OR-combined composite code-value-quantity filters, could not be executed on the HAPI FHIR server due to time-out errors starting on the smallest record count. Therefore, a simplified hemoglobin query variant was used for the server-to-server benchmark that only contained one of the OR-combined filters. As the HAPI FHIR server’s PostgreSQL database volume size exceeded the available disk space for the record count of 100,000, we excluded it from those benchmarks. The runs were repeated 10 times for the record counts, but only 3 times for the final 50,000 records due to the excessive runtime.</p>
      </sec>
      <sec>
        <title>Measuring Cache Impact</title>
        <p>The benchmarks so far were constructed to measure worst-case cold-start performance by repeatedly restarting the services between runs. To assess the impact of caching, we repeated the count and extracted workload benchmarks for the largest record count without restarting the services. The runs were repeated 5 times following 1 warm-up run.</p>
      </sec>
      <sec>
        <title>Measuring Data Skew Impact</title>
        <p>To evaluate robustness to realistic data skew, we ran another benchmark using three new count queries based on the distribution of “Observation” codes in the Synthea dataset for the largest record count: (1) skewed-hot codes, which filter on 5 high-frequency “Observation” codes selected from the 10 most common codes; (2) skewed-rare codes, which filter on 5 low-frequency “Observation” codes selected from the 10 rarest codes; and (3) skewed-mixed codes, which combine both the hot and rare codes using the logical OR operator.</p>
        <p>These workloads stress low-selectivity scans vs highly selective filters without modifying the dataset. Each query was executed 5 times, and we measured the mean of the execution time.</p>
      </sec>
    </sec>
    <sec sec-type="results">
      <title>Results</title>
      <sec>
        <title>Overview</title>
        <p><xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref> detail the number of resources corresponding to the Synthea record counts, the size of the raw JSON files, the time taken to import the resources to the FHIR servers and Delta Lake, and the size of the volumes used to store the imported resources.</p>
        <table-wrap position="float" id="table3">
          <label>Table 3</label>
          <caption>
            <p>Fast Healthcare Interoperability Resources (FHIR) resource counts, file size, import duration, and volume sizes corresponding to the Synthea record counts. The HAPI FHIR server was excluded from the largest record count benchmarks as the required disk space would exceed the available resources.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="110"/>
            <col width="110"/>
            <col width="130"/>
            <col width="140"/>
            <col width="130"/>
            <col width="130"/>
            <col width="0"/>
            <col width="100"/>
            <col width="150"/>
            <thead>
              <tr valign="top">
                <td>Records, n</td>
                <td colspan="6">FHIR resources by type, n</td>
                <td colspan="2">Synthea JSON file size</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Patient</td>
                <td>Condition</td>
                <td>Observation</td>
                <td>Encounter</td>
                <td>Total</td>
                <td colspan="2">Bulk</td>
                <td>Transaction</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>1000</td>
                <td>1144</td>
                <td>38,516</td>
                <td>515,968</td>
                <td>62,476</td>
                <td>618,104</td>
                <td colspan="2">653 MB</td>
                <td>1018 MB</td>
              </tr>
              <tr valign="top">
                <td>5000</td>
                <td>5760</td>
                <td>202,166</td>
                <td>2,963,729</td>
                <td>339,867</td>
                <td>3,511,522</td>
                <td colspan="2">3.6 GB</td>
                <td>5.6 GB</td>
              </tr>
              <tr valign="top">
                <td>10,000</td>
                <td>11,481</td>
                <td>402,577</td>
                <td>5,915,941</td>
                <td>683,151</td>
                <td>7,013,150</td>
                <td colspan="2">7.2 GB</td>
                <td>12 GB</td>
              </tr>
              <tr valign="top">
                <td>50,000</td>
                <td>57,538</td>
                <td>2,047,676</td>
                <td>30,371,746</td>
                <td>3,475,991</td>
                <td>35,952,951</td>
                <td colspan="2">37 GB</td>
                <td>57 GB</td>
              </tr>
              <tr valign="top">
                <td>100,000</td>
                <td>115,153</td>
                <td>4,075,593</td>
                <td>60,168,356</td>
                <td>6,925,962</td>
                <td>71,285,064</td>
                <td colspan="2">73 GB</td>
                <td>113 GB</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <table-wrap position="float" id="table4">
          <label>Table 4</label>
          <caption>
            <p>Import duration and volume sizes corresponding to the Synthea record counts. The HAPI Fast Healthcare Interoperability Resources (FHIR) server was excluded from the largest record count benchmarks as the required disk space would exceed available resources.</p>
          </caption>
          <table width="1000" cellpadding="5" cellspacing="0" border="1" rules="groups" frame="hsides">
            <col width="120"/>
            <col width="140"/>
            <col width="120"/>
            <col width="160"/>
            <col width="0"/>
            <col width="170"/>
            <col width="100"/>
            <col width="190"/>
            <thead>
              <tr valign="top">
                <td>Records, n</td>
                <td colspan="4">Import duration</td>
                <td colspan="3">Volume size</td>
              </tr>
              <tr valign="top">
                <td>
                  <break/>
                </td>
                <td>Pathling</td>
                <td>Blaze server</td>
                <td>HAPI server</td>
                <td colspan="2">MinIO (Pathling and Trino)</td>
                <td>Blaze</td>
                <td>HAPI PostgreSQL database</td>
              </tr>
            </thead>
            <tbody>
              <tr valign="top">
                <td>1000</td>
                <td>1 min 26 s</td>
                <td>2 min 56 s</td>
                <td>5 min 12 s</td>
                <td colspan="2">103.5 MB</td>
                <td>3.924 GB</td>
                <td>19.3 GB</td>
              </tr>
              <tr valign="top">
                <td>5000</td>
                <td>5 min 52 s</td>
                <td>14 min 52 s</td>
                <td>25 min 25 s</td>
                <td colspan="2">584.1 MB</td>
                <td>7.729 GB</td>
                <td>47.91 GB</td>
              </tr>
              <tr valign="top">
                <td>10,000</td>
                <td>10 min 50 s</td>
                <td>26 min 16 s</td>
                <td>52 min 23 s</td>
                <td colspan="2">1.177 GB</td>
                <td>11.33 GB</td>
                <td>78.53 GB</td>
              </tr>
              <tr valign="top">
                <td>50,000</td>
                <td>47 min 44 s</td>
                <td>2 h 12 min 49 s</td>
                <td>4 h 37 min 53 s</td>
                <td colspan="2">6.386 GB</td>
                <td>44.19 GB</td>
                <td>379.2 GB</td>
              </tr>
              <tr valign="top">
                <td>100,000</td>
                <td>1 h 26 min 27 s</td>
                <td>4 h 22 min 49 s</td>
                <td>N/A<sup>a</sup> (excluded)</td>
                <td colspan="2">11.35 GB</td>
                <td>84.05 GB</td>
                <td>N/A (excluded)</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="table4fn1">
              <p><sup>a</sup>N/A: not applicable.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Because of the way Synthea was implemented, specifying a record count did not exactly return the same number of “Patient” resources. In the following graphs, we refer to the dataset size as “record count,” corresponding to the individual resource counts in <xref ref-type="table" rid="table3">Tables 3</xref> and <xref ref-type="table" rid="table4">4</xref>.</p>
        <p><xref rid="figure2" ref-type="fig">Figures 2</xref> and <xref rid="figure3" ref-type="fig">3</xref> show the memory and CPU use during data loading across the evaluated systems. During the bulk import into the Pathling server, memory consumption rapidly reached the configured limits, and CPU use became saturated, particularly for larger record counts. After completion of the import, the execution of the Delta Lake OPTIMIZE and VACUUM commands resulted in a brief peak in both CPU and memory use due to the deletion of expired files from object storage.</p>
        <p>In contrast, the Blaze server exhibited lower CPU and memory use during transactional ingestion of FHIR resources. For the HAPI FHIR setup, CPU use was split between the PostgreSQL database and the server application, whereas memory consumption was dominated by the server. Following import, the execution of VACUUM and ANALYZE statements led to increased CPU and memory use on the PostgreSQL database, which was visible toward the end of the runtime.</p>
        <fig id="figure2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Central processing unit (CPU) use of the containers involved in loading the Fast Healthcare Interoperability Resources (FHIR) resources from the local file system into the servers and object storage.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig2.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Memory use of the containers involved in loading the Fast Healthcare Interoperability Resources (FHIR) resources from the local file system into the servers and object storage.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig3.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>FHIR Server Performance</title>
        <p><xref rid="figure4" ref-type="fig">Figure 4</xref> shows the duration of the count queries comparing Blaze and the HAPI FHIR server. For smaller record counts and the 2 simpler gender-age and diabetes queries, performance was similar. At the largest record counts, and across all queries, Blaze completed the requests faster. For the data extraction scenario shown in <xref rid="figure5" ref-type="fig">Figure 5</xref>, results varied depending on the query: Blaze exhibited lower execution time for gender-age extraction but performed worse for the diabetes query. Both servers demonstrated similar performance for the hemoglobin query. <xref rid="figure6" ref-type="fig">Figure 6</xref> shows the results of data aggregation queries. Across all record counts and queries, execution times were lower for Blaze than for HAPI FHIR.</p>
        <fig id="figure4" position="float">
          <label>Figure 4</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count for the HAPI Fast Healthcare Interoperability Resources and Blaze servers for count queries. Error bars indicate the 95% CIs of the mean. The P95 duration is indicated by a red cross. For the largest record count, runs were repeated 3 times; for all other record counts, runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig4.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count for the HAPI Fast Healthcare Interoperability Resources and Blaze servers for extraction queries. Error bars indicate the 95% CIs of the mean. The P95 duration is indicated by a red cross. For the largest record count, runs were repeated 3 times; for all other record counts, runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig5.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure6" position="float">
          <label>Figure 6</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count for the HAPI Fast Healthcare Interoperability Resources and Blaze servers for aggregation queries. Error bars indicate the 95% CIs of the mean. The P95 duration is indicated by a red cross. For the largest record count, runs were repeated 3 times; for all other record counts, runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig6.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Cross-Framework Performance Comparison</title>
        <p>For count workloads, Blaze exhibited subsecond execution times for the gender-age and diabetes queries across all record counts (<xref rid="figure7" ref-type="fig">Figure 7</xref>), outperforming both Trino and Pathling. For the complex hemoglobin query, Trino performed best for the largest record counts.</p>
        <fig id="figure7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count comparing Pathling, Blaze, and Trino for count queries. Error bars indicate the 95% CIs of the mean. All runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig7.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>For extraction workloads (<xref rid="figure8" ref-type="fig">Figure 8</xref>), Trino maintained the lowest mean latency for the largest record counts across all queries. FHIR-PYrate exhibited linear growth and substantially higher absolute latency at larger record counts, exceeding 1000 seconds at 100,000 records. At smaller sizes, it outperformed both Pathling and Trino. Pathling showed similar linear growth to that of FHIR-PYrate.</p>
        <fig id="figure8" position="float">
          <label>Figure 8</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count comparing Pathling, Blaze, and Trino for extraction queries. Error bars indicate the 95% CIs of the mean. All runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig8.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <p>Finally, the aggregate workload (<xref rid="figure9" ref-type="fig">Figure 9</xref>) amplified the performance differences observed in extraction queries. Trino again showed the lowest latency and the most stable scaling behavior across record counts. Pathling scaled linearly but with higher absolute runtimes than Trino. FHIR-PYrate showed the steepest performance degradation, with aggregate queries reaching multi-hour execution times at the largest record count.</p>
        <fig id="figure9" position="float">
          <label>Figure 9</label>
          <caption>
            <p>Mean execution times (log-scale; lower is better) by query and record count comparing Pathling, Blaze, and Trino for aggregation queries. Error bars indicate the 95% CIs of the mean. All runs were repeated 10 times.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig9.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Cache Impact</title>
        <p><xref rid="figure10" ref-type="fig">Figures 10</xref> and <xref rid="figure11" ref-type="fig">11</xref> show the impact of keeping services running for both the count (<xref rid="figure10" ref-type="fig">Figure 10</xref>) and extraction (<xref rid="figure11" ref-type="fig">Figure 11</xref>) query types. Cache warming had a small effect on count-based queries across all engines, whereas extraction workloads showed lower runtimes under warm-cache conditions for Pathling and Trino. The FHIR back end showed smaller differences between cold and warm execution.</p>
        <fig id="figure10" position="float">
          <label>Figure 10</label>
          <caption>
            <p>Log-scaled mean execution time for the count queries of 5 consecutive runs (lower is better) comparing warm-cache to cold-cache runtimes. Error bars indicate the 95% CIs of the mean.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig10.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
        <fig id="figure11" position="float">
          <label>Figure 11</label>
          <caption>
            <p>Log-scaled mean execution time for the extraction queries of 5 consecutive runs (lower is better) comparing warm-cache to cold-cache runtimes. Error bars indicate the 95% CIs of the mean.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig11.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
      <sec>
        <title>Data Skew</title>
        <p><xref rid="figure12" ref-type="fig">Figure 12</xref> shows execution times for count queries over hot, rare, and mixed “Observation” codes. The selected rare codes accounted for 0.0006% (394/60,168,356) of the total observations, and the hot codes accounted for 15.3% (9,188,358/60,168,356). Pathling showed little sensitivity to code frequency. Hot- and rare-code workloads exhibited comparable runtimes, indicating similar end-to-end execution costs across skew levels, with mixed-code queries taking the longest. Blaze showed strong sensitivity to skew. Rare-code queries were completed substantially faster than hot- and mixed-code workloads. Mixed-code performance was close to hot-code performance. Trino exhibited no sensitivity to skew, with all workloads performing nearly identically.</p>
        <fig id="figure12" position="float">
          <label>Figure 12</label>
          <caption>
            <p>Log-scaled mean execution time (lower is better) for counting “Observation” resources with common, rare, and either codes. Error bars indicate the 95% CIs of the mean.</p>
          </caption>
          <graphic xlink:href="medinform_v14i1e82924_fig12.png" alt-version="no" mimetype="image" position="float" xlink:type="simple"/>
        </fig>
      </sec>
    </sec>
    <sec sec-type="discussion">
      <title>Discussion</title>
      <p>There are at least 5 programming libraries that aim to implement analytical features using the FHIR Search API to retrieve data [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref11">11</xref>]. This is indicative not only of the increased use of FHIR as a format for conducting clinical research but also of an attempt at leveraging well-standardized interfaces: an FHIR server’s RESTful interface. However, the results of this benchmark show that frameworks purpose-built for analytics can execute common queries significantly faster. A key factor in this performance gap lies in the design differences between online transaction processing and online analytical processing systems. FHIR servers, typically designed for transactional workloads or accessing individual patient records, are not optimized for computationally intensive operations such as aggregations or large-scale data filtering and extraction. In contrast, Trino and Pathling leverage modern distributed data processing paradigms. These systems enable efficient execution plans by integrating data retrieval, computation, and caching mechanisms directly within their frameworks. Still, for the count workload, we found Blaze to outperform the tested alternatives for simpler queries and lower record counts.</p>
      <p>For the aggregation and extraction scenario, one performance-limiting factor was the need to fetch the results from the FHIR server sequentially using pagination, as the FHIR Search specification does not define native aggregation operations. The FHIR-PYrate implementation, therefore, was required to retrieve all matching “Observation” resources and perform grouping client side in pandas. In contrast, both Trino and Pathling executed the aggregation directly within the query engine using optimized query planning. The FHIR-PYrate library does allow for parallel fetching of search results, but this requires specifying an expected beginning and end date based on which to partition the search queries. For these scenarios, using the FHIR bulk export standard may be a higher-performing alternative [<xref ref-type="bibr" rid="ref28">28</xref>].</p>
      <p>We limited the selection of queries to only those that could be implemented as a single FHIR Search query. This avoids adding too much postprocessing logic to the total execution time. However, this also revealed challenges in the FHIR Search expressiveness and implementation in the server [<xref ref-type="bibr" rid="ref5">5</xref>]. For example, the diabetes query in the count scenario was initially supposed to count the number of distinct patients satisfying the criteria. This would require a count on the “Patient” resource type filtered using reverse chaining (“_has”) to only include those patients who were referenced in a diabetes “Condition” <italic>and</italic> in an “Encounter” in the specific date range within which the diagnosis was recorded. However, multiple “_has” parameters are processed independently, which would cause “Patient” resources with diabetes “Conditions” recorded as part of any “Encounter” to be counted as well. Furthermore, FHIR Search requires explicitly defining search parameters for nonstandard resource elements (such as for searching within extensions). Depending on the implementation, this also requires reindexing all resources before searches can be executed using custom parameters. Even the availability of standard search parameters varies between servers. Compared to FHIR Search, Trino is not as limiting in its expressiveness, as it is an American National Standards Institute SQL-compliant query engine. Similarly, Pathling is based on Apache Spark, where both SQL queries and a DataFrame API can be used to query the data.</p>
      <p>Both Pathling and Trino access the same data source—FHIR resources encoded as Delta Lake tables—but use different implementations to query them. This is an example of storage-compute decoupling, where storage and computation engines can be scaled and replaced independently. This also allows for querying the data using any other framework that supports reading Delta Lake tables and makes it easier for users to use tools they are already familiar with. Despite these advantages, the transition from FHIR-native data structures to tabular formats introduces challenges: encoding FHIR resources in Delta Lake tables requires additional infrastructure and processing steps. This process not only incurs extract, transform, and load overhead but also necessitates standardization efforts as the schema of these tables is not standardized and currently bound to the implementation of the Pathling encoders. Standardizing the encoding of FHIR resources as “Open” table formats might, therefore, be a useful future endeavor to ensure interoperability for analytics [<xref ref-type="bibr" rid="ref29">29</xref>]. Systematic performance benchmarking of alternative encodings of FHIR resources would be valuable future work as schema design choices can affect query performance across analytics engines.</p>
      <p>The resource use measured when loading the Synthea resources from the disk to the FHIR servers and Delta Lake tables differed across the tested system. Using Pathling, the FHIR JSON files were encoded faster and ended up using less disk space than both the HAPI and Blaze FHIR servers. However, Pathling also used more CPU and memory during loading. The 2 FHIR servers were loaded using 1 transaction bundle per Synthea patient record, and even though the blazectl client concurrency was set to 32, the servers did not fully saturate the available CPUs. This could be a result of the transaction processing implementation limiting the maximum concurrency. In comparison, the Pathling server used bulk import functionality, which was likely less affected by database locks and similar bottlenecks necessary to ensure atomic transaction processing. If it were implemented in all evaluated systems, consistently using bulk import would make the results more comparable.</p>
      <p>The limited cache sensitivity observed for count workloads suggests that these queries were dominated by full-table scans rather than metadata or plan reuse. In contrast, extraction workloads benefited from cache reuse in Pathling and Trino, likely reflecting reuse of Parquet or Delta Lake metadata, execution plans, and file system caches in Spark and MinIO. While Trino caches Delta Lake metadata by default (for 30 minutes), we did not enable file system caching, which might have shown a more significant impact on cache-warmed performance.</p>
      <p>The HAPI FHIR server was unable to complete the hemoglobin query beginning with the lowest record counts. When attempting to run the count variant of the query, we observed a single PostgreSQL CPU max out until eventually the query was aborted with an “out of disk space” error. We believe this may be caused by the way in which multiple OR-combined code-value-quantity searches are translated to SQL queries in the server.</p>
      <p>The observed differences in the impact of data skew on the count performance highlight how engines respond differently to skew in predicate selectivity and result cardinality. Pathling’s and Trino’s insensitivity to code frequency suggests that performance is dominated by full-table scans, with limited benefit from selective predicates in the current execution model. In contrast, Blaze benefited substantially from highly selective predicates. Mixed-code queries remained dominated by high-frequency codes, explaining why their performance resembled hot-code workloads. These results demonstrate that realistic long-tail code distributions can affect engine performance and that benchmarks assuming uniform code distributions may overstate the performance of scan-dominated pipelines. The present experiment isolated skew in single-table predicate selectivity; join-key skew and shuffle skew in multi-table and multi-worker cohort queries remain important directions for future work.</p>
      <p>All benchmarks were run on a single virtual machine, which means that both the benchmarked components and the code used to observe them were not isolated from each other. Additionally, all network communication occurred on the same local loop-back network interface, making the results less realistic as the impact of network latency was not measured. As modern data processing frameworks are capable of—and even optimized for—running analytical queries distributed across multiple compute nodes, how these findings generalize to such setups necessitates further exploration.</p>
      <p>Future research should aim to establish a common set of benchmark scenarios tailored to health care analytics on FHIR. Drawing inspiration from benchmarks such as TPC-H [<xref ref-type="bibr" rid="ref30">30</xref>], such a framework could provide a standardized methodology for evaluating diverse analytics engines and facilitate comparison across tools and architectures. This is particularly relevant with the new SQL-on-FHIR standard [<xref ref-type="bibr" rid="ref7">7</xref>].</p>
      <p>While our results show that FHIR Search performs poorly for analytical workloads dominated by large scans and aggregations, FHIR servers in general provide strengths that are critical in production environments: fine-grained access control; provenance and audit logging; transactional guarantees; and standardized functionality for searching, creating, updating, patching, and deletion. In many deployments, these features outweigh analytical throughput and justify the use of FHIR Search for patient-level queries and standardizable workflows [<xref ref-type="bibr" rid="ref31">31</xref>]. Our findings, therefore, do not argue against FHIR servers as a primary clinical data interface but, rather, highlight their unsuitability as a sole back end for large-scale cohort analytics and population-level queries, where columnar analytics engines can provide substantial performance advantages.</p>
      <p>The FHIR standard is primarily designed for interoperability and the exchange of health care data. In particular, while the REST-based FHIR Search API is useful for standard queries and retrieving specific patient records, it is generally not suitable for high-performance, complex analytics, making it necessary to move the data to a more robust analytical environment. Engines such as Trino and Pathling demonstrated substantial performance advantages for analytics scenarios.</p>
    </sec>
  </body>
  <back>
    <app-group/>
    <glossary>
      <title>Abbreviations</title>
      <def-list>
        <def-item>
          <term id="abb1">CPU</term>
          <def>
            <p>central processing unit</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb2">FHIR</term>
          <def>
            <p>Fast Healthcare Interoperability Resources</p>
          </def>
        </def-item>
        <def-item>
          <term id="abb3">REST</term>
          <def>
            <p>representational state transfer</p>
          </def>
        </def-item>
      </def-list>
    </glossary>
    <ack>
      <p>The present work was performed in (partial) fulfillment of the requirements for obtaining the degree “Dr. rer. biol. hum.” from the Friedrich-Alexander-Universität Erlangen-Nürnberg (JZ). During the preparation of this work, the authors used GPT-5.2 (OpenAI) to improve the readability and language of parts of the manuscript. After using this tool, the authors reviewed and edited the manuscript as needed and take full responsibility for the content of the published paper.</p>
    </ack>
    <notes>
      <sec>
        <title>Funding</title>
        <p>The present work was funded by the Bavarian Cancer Research Center. This publication was partially funded by the German Federal Ministry of Education and Research Network of University Medicine 3.0: “NUM 3.0” (grant 01KX2524, project NUM-DIZ).</p>
      </sec>
    </notes>
    <notes>
      <sec>
        <title>Data Availability</title>
        <p>The datasets generated during this study are available in the Zenodo repository [<xref ref-type="bibr" rid="ref32">32</xref>]. The analysis source code is available at GitHub [<xref ref-type="bibr" rid="ref33">33</xref>].</p>
      </sec>
    </notes>
    <fn-group>
      <fn fn-type="conflict">
        <p>JG is employed by the Australian eHealth Research Centre, which develops Pathling. JG was not involved in the design of the benchmark queries or the selection of workloads. All other authors declare no other conflicts of interest.</p>
      </fn>
    </fn-group>
    <ref-list>
      <ref id="ref1">
        <label>1</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Panagiotakos</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Real-world data: a brief review of the methods, applications, challenges and opportunities</article-title>
          <source>BMC Med Res Methodol</source>
          <year>2022</year>
          <month>11</month>
          <day>05</day>
          <volume>22</volume>
          <issue>1</issue>
          <fpage>287</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedresmethodol.biomedcentral.com/articles/10.1186/s12874-022-01768-6"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12874-022-01768-6</pub-id>
          <pub-id pub-id-type="medline">36335315</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12874-022-01768-6</pub-id>
          <pub-id pub-id-type="pmcid">PMC9636688</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref2">
        <label>2</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Rubinstein</surname>
              <given-names>SM</given-names>
            </name>
            <name name-style="western">
              <surname>Nead</surname>
              <given-names>KT</given-names>
            </name>
            <name name-style="western">
              <surname>Wojcieszynski</surname>
              <given-names>AP</given-names>
            </name>
            <name name-style="western">
              <surname>Gabriel</surname>
              <given-names>PE</given-names>
            </name>
            <name name-style="western">
              <surname>Warner</surname>
              <given-names>JL</given-names>
            </name>
          </person-group>
          <article-title>The evolving use of electronic health records (EHR) for research</article-title>
          <source>Semin Radiat Oncol</source>
          <year>2019</year>
          <month>10</month>
          <volume>29</volume>
          <issue>4</issue>
          <fpage>354</fpage>
          <lpage>61</lpage>
          <pub-id pub-id-type="doi">10.1016/j.semradonc.2019.05.010</pub-id>
          <pub-id pub-id-type="medline">31472738</pub-id>
          <pub-id pub-id-type="pii">S1053-4296(19)30042-6</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref3">
        <label>3</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Fu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Leung</surname>
              <given-names>LY</given-names>
            </name>
            <name name-style="western">
              <surname>Raulli</surname>
              <given-names>AO</given-names>
            </name>
            <name name-style="western">
              <surname>Kallmes</surname>
              <given-names>DF</given-names>
            </name>
            <name name-style="western">
              <surname>Kinsman</surname>
              <given-names>KA</given-names>
            </name>
            <name name-style="western">
              <surname>Nelson</surname>
              <given-names>KB</given-names>
            </name>
            <name name-style="western">
              <surname>Clark</surname>
              <given-names>MS</given-names>
            </name>
            <name name-style="western">
              <surname>Luetmer</surname>
              <given-names>PH</given-names>
            </name>
            <name name-style="western">
              <surname>Kingsbury</surname>
              <given-names>PR</given-names>
            </name>
            <name name-style="western">
              <surname>Kent</surname>
              <given-names>DM</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <article-title>Assessment of the impact of EHR heterogeneity for clinical research through a case study of silent brain infarction</article-title>
          <source>BMC Med Inform Decis Mak</source>
          <year>2020</year>
          <month>03</month>
          <day>30</day>
          <volume>20</volume>
          <issue>1</issue>
          <fpage>60</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmcmedinformdecismak.biomedcentral.com/articles/10.1186/s12911-020-1072-9"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12911-020-1072-9</pub-id>
          <pub-id pub-id-type="medline">32228556</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12911-020-1072-9</pub-id>
          <pub-id pub-id-type="pmcid">PMC7106829</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref4">
        <label>4</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Duda</surname>
              <given-names>SN</given-names>
            </name>
            <name name-style="western">
              <surname>Kennedy</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Conway</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Cheng</surname>
              <given-names>AC</given-names>
            </name>
            <name name-style="western">
              <surname>Nguyen</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Zayas-Cabán</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Harris</surname>
              <given-names>PA</given-names>
            </name>
          </person-group>
          <article-title>HL7 FHIR-based tools and initiatives to support clinical research: a scoping review</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2022</year>
          <month>08</month>
          <day>16</day>
          <volume>29</volume>
          <issue>9</issue>
          <fpage>1642</fpage>
          <lpage>53</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/35818340"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocac105</pub-id>
          <pub-id pub-id-type="medline">35818340</pub-id>
          <pub-id pub-id-type="pii">6639865</pub-id>
          <pub-id pub-id-type="pmcid">PMC9382376</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref5">
        <label>5</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gulden</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Mate</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Prokosch</surname>
              <given-names>HU</given-names>
            </name>
            <name name-style="western">
              <surname>Kraus</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Investigating the capabilities of FHIR search for clinical trial phenotyping</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2018</year>
          <volume>253</volume>
          <fpage>3</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="medline">30147028</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref6">
        <label>6</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ziegler</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Erpenbeck</surname>
              <given-names>MP</given-names>
            </name>
            <name name-style="western">
              <surname>Fuchs</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Saibold</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Volkmer</surname>
              <given-names>PC</given-names>
            </name>
            <name name-style="western">
              <surname>Schmidt</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Eicher</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Pallaoro</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>De Souza Falguera</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Aubele</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hagedorn</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Vansovich</surname>
              <given-names>E</given-names>
            </name>
            <name name-style="western">
              <surname>Raffler</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ringshandl</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Kerscher</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Maurer</surname>
              <given-names>JK</given-names>
            </name>
            <name name-style="western">
              <surname>Kühnel</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Schenkirsch</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kampf</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Kapsner</surname>
              <given-names>LA</given-names>
            </name>
            <name name-style="western">
              <surname>Ghanbarian</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Spengler</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Soto-Rey</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Albashiti</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Hellwig</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Ertl</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fette</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Kraska</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Boeker</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Prokosch</surname>
              <given-names>HU</given-names>
            </name>
            <name name-style="western">
              <surname>Gulden</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Bridging data silos in oncology with modular software for federated analysis on fast healthcare interoperability resources: multisite implementation study</article-title>
          <source>J Med Internet Res</source>
          <year>2025</year>
          <month>04</month>
          <day>15</day>
          <volume>27</volume>
          <fpage>e65681</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2025//e65681/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/65681</pub-id>
          <pub-id pub-id-type="medline">40233352</pub-id>
          <pub-id pub-id-type="pii">v27i1e65681</pub-id>
          <pub-id pub-id-type="pmcid">PMC12041822</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref7">
        <label>7</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grimes</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Brush</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Rhyzhikov</surname>
              <given-names>N</given-names>
            </name>
            <name name-style="western">
              <surname>Szul</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Mandel</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gottlieb</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Grieve</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Sadjad</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Sanyal</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>SQL on FHIR - tabular views of FHIR data using FHIRPath</article-title>
          <source>NPJ Digit Med</source>
          <year>2025</year>
          <month>06</month>
          <day>09</day>
          <volume>8</volume>
          <issue>1</issue>
          <fpage>342</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://doi.org/10.1038/s41746-025-01708-w"/>
          </comment>
          <pub-id pub-id-type="doi">10.1038/s41746-025-01708-w</pub-id>
          <pub-id pub-id-type="medline">40490535</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-025-01708-w</pub-id>
          <pub-id pub-id-type="pmcid">PMC12149319</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref8">
        <label>8</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Oehm</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Storck</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Fechner</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Brix</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Yildirim</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Dugas</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>FhirExtinguisher: a FHIR resource flattening tool using FHIRPath</article-title>
          <source>Stud Health Technol Inform</source>
          <year>2021</year>
          <month>05</month>
          <day>27</day>
          <volume>281</volume>
          <fpage>1112</fpage>
          <lpage>3</lpage>
          <pub-id pub-id-type="doi">10.3233/SHTI210369</pub-id>
          <pub-id pub-id-type="medline">34042862</pub-id>
          <pub-id pub-id-type="pii">SHTI210369</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref9">
        <label>9</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hosch</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Baldini</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Parmar</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Borys</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Koitka</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Engelke</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Arzideh</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Ulrich</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Nensa</surname>
              <given-names>F</given-names>
            </name>
          </person-group>
          <article-title>FHIR-PYrate: a data science friendly Python package to query FHIR servers</article-title>
          <source>BMC Health Serv Res</source>
          <year>2023</year>
          <month>07</month>
          <day>06</day>
          <volume>23</volume>
          <issue>1</issue>
          <fpage>734</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://bmchealthservres.biomedcentral.com/articles/10.1186/s12913-023-09498-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s12913-023-09498-1</pub-id>
          <pub-id pub-id-type="medline">37415138</pub-id>
          <pub-id pub-id-type="pii">10.1186/s12913-023-09498-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC10326955</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref10">
        <label>10</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Palm</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Meineke</surname>
              <given-names>FA</given-names>
            </name>
            <name name-style="western">
              <surname>Przybilla</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Peschel</surname>
              <given-names>T</given-names>
            </name>
          </person-group>
          <article-title>"fhircrackr": an R package unlocking fast healthcare interoperability resources for statistical analysis</article-title>
          <source>Appl Clin Inform</source>
          <year>2023</year>
          <month>01</month>
          <volume>14</volume>
          <issue>1</issue>
          <fpage>54</fpage>
          <lpage>64</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="http://www.thieme-connect.com/DOI/DOI?10.1055/s-0042-1760436"/>
          </comment>
          <pub-id pub-id-type="doi">10.1055/s-0042-1760436</pub-id>
          <pub-id pub-id-type="medline">36696915</pub-id>
          <pub-id pub-id-type="pmcid">PMC9876659</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref11">
        <label>11</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Brehmer</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Sauer</surname>
              <given-names>CM</given-names>
            </name>
            <name name-style="western">
              <surname>Salazar Rodríguez</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Herrmann</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Kim</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Keyl</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bahnsen</surname>
              <given-names>FH</given-names>
            </name>
            <name name-style="western">
              <surname>Frank</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Köhrmann</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Rassaf</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mahabadi</surname>
              <given-names>AA</given-names>
            </name>
            <name name-style="western">
              <surname>Hadaschik</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Darr</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Herrmann</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Buer</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Brenner</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Reinhardt</surname>
              <given-names>HC</given-names>
            </name>
            <name name-style="western">
              <surname>Nensa</surname>
              <given-names>F</given-names>
            </name>
            <name name-style="western">
              <surname>Gertz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Egger</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kleesiek</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Establishing medical intelligence-leveraging Fast Healthcare Interoperability Resources to improve clinical management: retrospective cohort and clinical implementation study</article-title>
          <source>J Med Internet Res</source>
          <year>2024</year>
          <month>10</month>
          <day>31</day>
          <volume>26</volume>
          <fpage>e55148</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.jmir.org/2024//e55148/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/55148</pub-id>
          <pub-id pub-id-type="medline">39240144</pub-id>
          <pub-id pub-id-type="pii">v26i1e55148</pub-id>
          <pub-id pub-id-type="pmcid">PMC11565078</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref12">
        <label>12</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grimes</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Szul</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Metke-Jimenez</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Lawley</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Loi</surname>
              <given-names>K</given-names>
            </name>
          </person-group>
          <article-title>Pathling: analytics on FHIR</article-title>
          <source>J Biomed Semantics</source>
          <year>2022</year>
          <month>09</month>
          <day>08</day>
          <volume>13</volume>
          <issue>1</issue>
          <fpage>23</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://jbiomedsem.biomedcentral.com/articles/10.1186/s13326-022-00277-1"/>
          </comment>
          <pub-id pub-id-type="doi">10.1186/s13326-022-00277-1</pub-id>
          <pub-id pub-id-type="medline">36076268</pub-id>
          <pub-id pub-id-type="pii">10.1186/s13326-022-00277-1</pub-id>
          <pub-id pub-id-type="pmcid">PMC9455941</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref13">
        <label>13</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gruendner</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Gulden</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Kampf</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Mate</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Prokosch</surname>
              <given-names>HU</given-names>
            </name>
            <name name-style="western">
              <surname>Zierk</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A framework for criteria-based selection and processing of Fast Healthcare Interoperability Resources (FHIR) data for statistical analysis: design and implementation study</article-title>
          <source>JMIR Med Inform</source>
          <year>2021</year>
          <month>04</month>
          <day>01</day>
          <volume>9</volume>
          <issue>4</issue>
          <fpage>e25645</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://medinform.jmir.org/2021/4/e25645/"/>
          </comment>
          <pub-id pub-id-type="doi">10.2196/25645</pub-id>
          <pub-id pub-id-type="medline">33792554</pub-id>
          <pub-id pub-id-type="pii">v9i4e25645</pub-id>
          <pub-id pub-id-type="pmcid">PMC8050750</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref14">
        <label>14</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ayaz</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Pasha</surname>
              <given-names>MF</given-names>
            </name>
            <name name-style="western">
              <surname>Alahmadi</surname>
              <given-names>TJ</given-names>
            </name>
            <name name-style="western">
              <surname>Abdullah</surname>
              <given-names>NN</given-names>
            </name>
            <name name-style="western">
              <surname>Alkahtani</surname>
              <given-names>HK</given-names>
            </name>
          </person-group>
          <article-title>Transforming healthcare analytics with FHIR: a framework for standardizing and analyzing clinical data</article-title>
          <source>Healthcare (Basel)</source>
          <year>2023</year>
          <month>06</month>
          <day>13</day>
          <volume>11</volume>
          <issue>12</issue>
          <fpage>1729</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://www.mdpi.com/resolver?pii=healthcare11121729"/>
          </comment>
          <pub-id pub-id-type="doi">10.3390/healthcare11121729</pub-id>
          <pub-id pub-id-type="medline">37372847</pub-id>
          <pub-id pub-id-type="pii">healthcare11121729</pub-id>
          <pub-id pub-id-type="pmcid">PMC10298100</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref15">
        <label>15</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Sahu</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Ignatov</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Gottlieb</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Mandl</surname>
              <given-names>KD</given-names>
            </name>
          </person-group>
          <article-title>High performance computing on Flat FHIR files created with the new SMART/HL7 Bulk Data access standard</article-title>
          <source>AMIA Annu Symp Proc</source>
          <year>2020</year>
          <month>03</month>
          <day>04</day>
          <volume>2019</volume>
          <fpage>592</fpage>
          <lpage>6</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/32308853"/>
          </comment>
          <pub-id pub-id-type="medline">32308853</pub-id>
          <pub-id pub-id-type="pmcid">PMC7153160</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref16">
        <label>16</label>
        <nlm-citation citation-type="web">
          <article-title>FHIRPath (normative release)</article-title>
          <source>FHIR</source>
          <access-date>2024-08-08</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://hl7.org/fhirpath/">https://hl7.org/fhirpath/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref17">
        <label>17</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zaharia</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Xin</surname>
              <given-names>RS</given-names>
            </name>
            <name name-style="western">
              <surname>Wendell</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Das</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Armbrust</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Dave</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Meng</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Rosen</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Venkataraman</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Franklin</surname>
              <given-names>MJ</given-names>
            </name>
            <name name-style="western">
              <surname>Ghodsi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Gonzalez</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Shenker</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Stoica</surname>
              <given-names>I</given-names>
            </name>
          </person-group>
          <article-title>Apache Spark: a unified engine for big data processing</article-title>
          <source>Commun ACM</source>
          <year>2016</year>
          <month>10</month>
          <day>28</day>
          <volume>59</volume>
          <issue>11</issue>
          <fpage>56</fpage>
          <lpage>65</lpage>
          <pub-id pub-id-type="doi">10.1145/2934664</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref18">
        <label>18</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zhao</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Zou</surname>
              <given-names>Q</given-names>
            </name>
            <name name-style="western">
              <surname>Fang</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Peng</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Bioinformatics applications on Apache Spark</article-title>
          <source>Gigascience</source>
          <year>2018</year>
          <month>08</month>
          <day>01</day>
          <volume>7</volume>
          <issue>8</issue>
          <fpage>giy098</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/30101283"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/gigascience/giy098</pub-id>
          <pub-id pub-id-type="medline">30101283</pub-id>
          <pub-id pub-id-type="pii">5067872</pub-id>
          <pub-id pub-id-type="pmcid">PMC6113509</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref19">
        <label>19</label>
        <nlm-citation citation-type="web">
          <source>Trino</source>
          <access-date>2024-08-08</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://trino.io/">https://trino.io/</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref20">
        <label>20</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Röhm</surname>
              <given-names>U</given-names>
            </name>
            <name name-style="western">
              <surname>Brent</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Dawborn</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Jeffries</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>SQL for data scientists: designing SQL tutorials for scalable online teaching</article-title>
          <source>Proc VLDB Endow</source>
          <year>2020</year>
          <month>08</month>
          <day>1</day>
          <volume>13</volume>
          <issue>12</issue>
          <fpage>2989</fpage>
          <lpage>92</lpage>
          <pub-id pub-id-type="doi">10.14778/3415478.3415526</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref21">
        <label>21</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Armbrust</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Das</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>L</given-names>
            </name>
            <name name-style="western">
              <surname>Yavuz</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Zhu</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Murthy</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Torres</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>van Hovell</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Ionescu</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Łuszczak</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Świtakowski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Szafrański</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name name-style="western">
              <surname>Ueshin</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Mokhtar</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Boncz</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Ghodsi</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Paranjpye</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Senster</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Xin</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Zaharia</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>Delta lake: high-performance ACID table storage over cloud object stores</article-title>
          <source>Proc VLDB Endow</source>
          <year>2020</year>
          <month>08</month>
          <day>1</day>
          <volume>13</volume>
          <issue>12</issue>
          <fpage>3411</fpage>
          <lpage>24</lpage>
          <pub-id pub-id-type="doi">10.14778/3415478.3415560</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref22">
        <label>22</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dageville</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Cruanes</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Zukowski</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Antonov</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Avanes</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Bock</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Claybaugh</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Engovatov</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Hentschel</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Lee</surname>
              <given-names>AW</given-names>
            </name>
            <name name-style="western">
              <surname>Motivala</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Munir</surname>
              <given-names>AQ</given-names>
            </name>
            <name name-style="western">
              <surname>Pelley</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Povinec</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Rahn</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Triantafyllis</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Unterbrunner</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <article-title>The Snowflake Elastic Data Warehouse</article-title>
          <source>Proceedings of the 2016 International Conference on Management of Data</source>
          <year>2016</year>
          <conf-name>SIGMOD '16</conf-name>
          <conf-date>Jun 26-Jul 1, 2016</conf-date>
          <conf-loc>San Francisco, CA</conf-loc>
          <pub-id pub-id-type="doi">10.1145/2882903.2903741</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref23">
        <label>23</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Bai</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Wang</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Xu</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Towards Elastic Data Warehousing by decoupling data management and computation</article-title>
          <source>Proceedings of the 2020 4th International Conference on Cloud and Big Data Computing</source>
          <year>2020</year>
          <conf-name>ICCBDC '20</conf-name>
          <conf-date>Aug 26-28, 2020</conf-date>
          <conf-loc>Virtual Event</conf-loc>
          <pub-id pub-id-type="doi">10.1145/3416921.3416935</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref24">
        <label>24</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Walonoski</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Kramer</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Nichols</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Quina</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Moesel</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Hall</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Duffett</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Dube</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Gallagher</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>McLachlan</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Synthea: an approach, method, and software mechanism for generating synthetic patients and the synthetic electronic health care record</article-title>
          <source>J Am Med Inform Assoc</source>
          <year>2018</year>
          <month>03</month>
          <day>01</day>
          <volume>25</volume>
          <issue>3</issue>
          <fpage>230</fpage>
          <lpage>8</lpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://europepmc.org/abstract/MED/29025144"/>
          </comment>
          <pub-id pub-id-type="doi">10.1093/jamia/ocx079</pub-id>
          <pub-id pub-id-type="medline">29025144</pub-id>
          <pub-id pub-id-type="pii">4098271</pub-id>
          <pub-id pub-id-type="pmcid">PMC7651916</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref25">
        <label>25</label>
        <nlm-citation citation-type="confproc">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Zhuang</surname>
              <given-names>Z</given-names>
            </name>
            <name name-style="western">
              <surname>Tran</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Weng</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ramachandra</surname>
              <given-names>H</given-names>
            </name>
            <name name-style="western">
              <surname>Sridharan</surname>
              <given-names>B</given-names>
            </name>
          </person-group>
          <article-title>Taming memory related performance pitfalls in Linux Cgroups</article-title>
          <source>Proceedings of the 2017 International Conference on Computing, Networking and Communications</source>
          <year>2017</year>
          <conf-name>ICNC 2017</conf-name>
          <conf-date>Jan 26-29, 2017</conf-date>
          <conf-loc>Silicon Valley, CA</conf-loc>
          <pub-id pub-id-type="doi">10.1109/iccnc.2017.7876184</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref26">
        <label>26</label>
        <nlm-citation citation-type="web">
          <article-title>Production configuration guide</article-title>
          <source>Blaze</source>
          <access-date>2026-03-01</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://samply.github.io/blaze/production-configuration.html">https://samply.github.io/blaze/production-configuration.html</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref27">
        <label>27</label>
        <nlm-citation citation-type="web">
          <article-title>le0pard/pgtune</article-title>
          <source>GitHub</source>
          <access-date>2026-03-01</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/le0pard/pgtune">https://github.com/le0pard/pgtune</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref28">
        <label>28</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mandl</surname>
              <given-names>KD</given-names>
            </name>
            <name name-style="western">
              <surname>Gottlieb</surname>
              <given-names>D</given-names>
            </name>
            <name name-style="western">
              <surname>Mandel</surname>
              <given-names>JC</given-names>
            </name>
            <name name-style="western">
              <surname>Ignatov</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>Sayeed</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Grieve</surname>
              <given-names>G</given-names>
            </name>
            <name name-style="western">
              <surname>Jones</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Ellis</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Culbertson</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Push button population health: the SMART/HL7 FHIR Bulk Data Access application programming interface</article-title>
          <source>NPJ Digit Med</source>
          <year>2020</year>
          <month>11</month>
          <day>19</day>
          <volume>3</volume>
          <issue>1</issue>
          <fpage>151</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-020-00358-4</pub-id>
          <pub-id pub-id-type="medline">33299056</pub-id>
          <pub-id pub-id-type="pii">10.1038/s41746-020-00358-4</pub-id>
          <pub-id pub-id-type="pmcid">PMC7678833</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref29">
        <label>29</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Grimes</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>aehrc/parquet-on-fhir: v0.1</article-title>
          <source>Zenodo</source>
          <year>2024</year>
          <month>11</month>
          <day>15</day>
          <access-date>2025-01-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://zenodo.org/records/14166591">https://zenodo.org/records/14166591</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref30">
        <label>30</label>
        <nlm-citation citation-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Barata</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Bernardino</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Furtado</surname>
              <given-names>P</given-names>
            </name>
          </person-group>
          <person-group person-group-type="editor">
            <name name-style="western">
              <surname>Rocha</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Correia</surname>
              <given-names>A</given-names>
            </name>
            <name name-style="western">
              <surname>Costanzo</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Reis</surname>
              <given-names>L</given-names>
            </name>
          </person-group>
          <article-title>An overview of decision support benchmarks: TPC-DS, TPC-HSSB</article-title>
          <source>New Contributions in Information Systems and Technologies</source>
          <year>2015</year>
          <publisher-loc>Cham, Switzerland</publisher-loc>
          <publisher-name>Springer</publisher-name>
        </nlm-citation>
      </ref>
      <ref id="ref31">
        <label>31</label>
        <nlm-citation citation-type="journal">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gulden</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Macho</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Reinecke</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Strantz</surname>
              <given-names>C</given-names>
            </name>
            <name name-style="western">
              <surname>Prokosch</surname>
              <given-names>HU</given-names>
            </name>
            <name name-style="western">
              <surname>Blasini</surname>
              <given-names>R</given-names>
            </name>
          </person-group>
          <article-title>recruIT: a cloud-native clinical trial recruitment support system based on Health Level 7 Fast Healthcare Interoperability Resources (HL7 FHIR) and the Observational Medical Outcomes Partnership Common Data Model (OMOP CDM)</article-title>
          <source>Comput Biol Med</source>
          <year>2024</year>
          <month>05</month>
          <volume>174</volume>
          <fpage>108411</fpage>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://linkinghub.elsevier.com/retrieve/pii/S0010-4825(24)00495-5"/>
          </comment>
          <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.108411</pub-id>
          <pub-id pub-id-type="medline">38626510</pub-id>
          <pub-id pub-id-type="pii">S0010-4825(24)00495-5</pub-id>
        </nlm-citation>
      </ref>
      <ref id="ref32">
        <label>32</label>
        <nlm-citation citation-type="web">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gulden</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>Benchmark run results comparing the performance of the Trino, Pathling, and FHIR-PYrate query engines for large-scale analytics of FHIR resources</article-title>
          <source>Zenodo</source>
          <year>2026</year>
          <month>01</month>
          <day>01</day>
          <access-date>2026-07-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://zenodo.org/records/18827363">https://zenodo.org/records/18827363</ext-link>
          </comment>
        </nlm-citation>
      </ref>
      <ref id="ref33">
        <label>33</label>
        <nlm-citation citation-type="web">
          <article-title>bzkf / analytics-on-fhir-benchmark</article-title>
          <source>GitHub</source>
          <access-date>2026-07-07</access-date>
          <comment>
            <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://github.com/bzkf/analytics-on-fhir-benchmark">https://github.com/bzkf/analytics-on-fhir-benchmark</ext-link>
          </comment>
        </nlm-citation>
      </ref>
    </ref-list>
  </back>
</article>
