<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e88390</article-id><article-id pub-id-type="doi">10.2196/88390</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Privacy Leakage in Federated Learning in Radiology Reports: Comparative Evaluation of Tokenizer and Batch-Size Privacy Risks</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Parampottupadam</surname><given-names>Santhosh</given-names></name><degrees>MSc, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Mart&#x00ED;nez Mora</surname><given-names>Andr&#x00E9;s</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Bounias</surname><given-names>Dimitrios</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sav</surname><given-names>Sinem</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Maier-Hein</surname><given-names>Klaus</given-names></name><degrees>Prof Dr</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Floca</surname><given-names>Ralf</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Division of Medical Image Computing, German Cancer Research Center</institution><addr-line>Im Neuenheimer Feld 280</addr-line><addr-line>Heidelberg</addr-line><addr-line>Baden-Wurttemberg</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Medical Faculty Heidelberg, Heidelberg University</institution><addr-line>Heidelberg</addr-line><country>Germany</country></aff><aff id="aff3"><institution>Department of Computer Engineering, Bilkent University</institution><addr-line>Ankara</addr-line><country>T&#x00FC;rkiye</country></aff><aff id="aff4"><institution>Pattern Analysis and Learning Group, Department of Radiation Oncology, Heidelberg University Hospital</institution><addr-line>Heidelberg</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Goyal</surname><given-names>Nitin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Taiwo</surname><given-names>Peter</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Santhosh Parampottupadam, MSc, PhD, Division of Medical Image Computing, German Cancer Research Center, Im Neuenheimer Feld 280, Heidelberg, Baden-Wurttemberg, 69120, Germany, 49 06221 420; <email>santhosh.parampottupadam@dkfz-heidelberg.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>11</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e88390</elocation-id><history><date date-type="received"><day>24</day><month>11</month><year>2025</year></date><date date-type="rev-recd"><day>21</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>23</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Santhosh Parampottupadam, Andr&#x00E9;s Mart&#x00ED;nez Mora, Dimitrios Bounias, Sinem Sav, Klaus Maier-Hein, Ralf Floca. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 11.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e88390"/><abstract><sec><title>Background</title><p>Federated learning (FL) enables multi-institutional model training on clinical text without sharing raw data; however, gradient inversion methods can reconstruct sensitive information from shared model updates. The extent of such privacy leakage in FL applied to radiology reports, and the role of tokenizer design, remains unclear.</p></sec><sec><title>Objective</title><p>This study aimed to quantify gradient-based reconstruction of radiology report text in an FL setting and to compare privacy risk across 3 transformer tokenization strategies in a controlled, tokenizer-aware evaluation.</p></sec><sec sec-type="methods"><title>Methods</title><p>Six FL clients trained a GPT-2&#x2013;style transformer (sequence length 32) on 2 public clinical-text corpora comprising 368,751 diagnostic reports, 98,206 discharge summaries, and 1500 MIMIC-CXR (Medical Information Mart for Intensive Care Chest X-Ray) radiology reports. Models were trained using 3 tokenizers (GPT-2, RadBERT, and LLaMA-2) with batch sizes of 64, 128, and 256. An active malicious-server threat model was assumed, and analytic gradient inversion was applied to recover text. Reconstruction fidelity was measured over 5 runs using exact sentence accuracy, sentence-level bilingual evaluation understudy (S-BLEU), and recall-oriented understudy for gisting evaluation (ROUGE-L).</p></sec><sec sec-type="results"><title>Results</title><p>Exact sentence reconstruction ranged from 27% to 75% across tokenizers, datasets, and batch sizes. At batch size 64 on the discharge dataset, accuracy was 64.7% (GPT-2), 70% (RadBERT), and 67.5% (LLaMA-2), decreasing to 27.3%, 28.5%, and 27.5% at batch size 256. S-BLEU declined with increasing batch size (eg, discharge reports from 0.69 to 0.31). Reconstruction fidelity did not differ significantly across tokenizers (all but one of 27 comparisons nonsignificant; none significant after Holm correction), and approximately 75% of clinical concepts were represented in the reconstructed text (a corpus-level upper bound), regardless of tokenizer. Batch size was the dominant factor governing leakage.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Under a worst-case malicious server that tampers with the shared model and observes unprotected per-client gradients (no secure aggregation or differential privacy), substantial portions of radiology-report text can be reconstructed, with up to approximately 75% of reconstructed 32-token sequences and 75% of clinical concepts (not direct patient identifiers) recovered from FL gradients. In a controlled ablation holding model architecture fixed, tokenizer choice, including domain-specific tokenizers, did not significantly affect leakage under the evaluated conditions, whereas batch size was the primary determinant, and no tokenizer significantly reduced the risk. Tokenizer selection should therefore not be treated as a privacy safeguard in this setting. Safeguards such as secure aggregation and differential privacy should therefore be evaluated as candidate protections for FL deployments that must satisfy Health Insurance Portability and Accountability Act (HIPAA) and General Data Protection Regulation (GDPR) requirements in radiology natural language processing (NLP); legal compliance additionally depends on organizational safeguards, risk assessment, and governance beyond the scope of this study.</p></sec></abstract><kwd-group><kwd>federated learning</kwd><kwd>radiology</kwd><kwd>privacy</kwd><kwd>gradient inversion</kwd><kwd>large language models</kwd><kwd>data security</kwd><kwd>transformer models</kwd><kwd>patient confidentiality</kwd><kwd>Health Insurance Portability and Accountability Act</kwd><kwd>HIPAA</kwd><kwd>General Data Protection Regulation</kwd><kwd>GDPR</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Radiology reports serve as an indispensable tool in medical diagnostics, providing critical insights [<xref ref-type="bibr" rid="ref1">1</xref>] that complement imaging data. These textual data contain rich clinical information, including patient history, diagnostic impressions, and recommended follow-ups, often bridging gaps left by imaging alone. In recent years, advancements in AI, particularly transformer-based large language models (LLMs) [<xref ref-type="bibr" rid="ref2">2</xref>], have revolutionized natural language processing (NLP), making it possible to analyze unstructured textual data at scale. LLMs such as the GPT family [<xref ref-type="bibr" rid="ref3">3</xref>] are now being explored for their potential in radiology to assist in generating summaries [<xref ref-type="bibr" rid="ref4">4</xref>], extracting relevant clinical insights, and identifying patterns across large datasets [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. A recent systematic review of LLM evaluations in clinical medicine highlights the rapid growth of these models and underscores the need for robust evaluation frameworks to ensure their safety, reliability, and ethical alignment in health care applications [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Radiology LLMs can be collaboratively developed across multiple institutions, leveraging diverse clinical reports to enhance model robustness and improve generalizability across unseen health care settings. Nevertheless, health care data, such as radiological data sharing, faces significant challenges due to stringent data protection laws such as the Health Insurance Portability and Accountability Act (HIPAA) [<xref ref-type="bibr" rid="ref9">9</xref>] in the United States and the General Data Protection Regulation (GDPR) [<xref ref-type="bibr" rid="ref10">10</xref>] in the European Union, which mandate strict safeguards for the privacy of sensitive patient data. Federated learning (FL) [<xref ref-type="bibr" rid="ref11">11</xref>], a decentralized machine learning approach where models are trained across multiple institutions without exchanging raw data, has emerged as a promising solution to address these challenges by ensuring that sensitive data remain within each participating institution while still enabling collaborative model development. Multicentric FL clinical collaborations [<xref ref-type="bibr" rid="ref12">12</xref>-<xref ref-type="bibr" rid="ref15">15</xref>] are transforming medical research by enabling privacy-preserving data sharing, fostering innovation, and addressing critical concerns around data ownership and security.</p><p>Nevertheless, FL&#x2019;s decentralized nature introduces inherent vulnerabilities that may compromise patient confidentiality [<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. While they eliminate direct data sharing, FL systems rely on the exchange of model parameters between institutions and a central server. An active malicious server&#x2014;one that not only observes gradients but also modifies the shared model architecture before distribution&#x2014;can exploit these parameters to reconstruct sensitive data, exposing significant vulnerabilities. Existing research has shown that gradient-inversion attacks [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>] can reconstruct private text from models trained on generic language datasets, but such risks remain largely unexplored in the domain of radiology [<xref ref-type="bibr" rid="ref21">21</xref>].</p><p>Recent advances in NLP have demonstrated that leveraging domain-specific tokenizers [<xref ref-type="bibr" rid="ref18">18</xref>] and training LLMs on domain-adapted corpora significantly enhances performance in specialized fields such as medicine and genomics. For instance, studies such as Gu et al [<xref ref-type="bibr" rid="ref22">22</xref>] introduced PubMedBERT, showing that pretraining on biomedical literature yields superior results on downstream clinical tasks compared to generalist models. Similarly, prior studies such as Zhang et al [<xref ref-type="bibr" rid="ref23">23</xref>] demonstrate that fine-tuning LLMs such as BioBERT and MedAlpaca with domain-specific data and tailored tokenization strategies substantially enhances performance in medical tasks. These adaptations enable the models to capture nuanced, domain-relevant semantics that general-purpose LLMs typically overlook.</p><p>To specifically investigate the role of tokenization in privacy leakage within FL, we isolate tokenizer design as the primary variable while keeping the underlying model architecture constant. While most prior work compares entire language models [<xref ref-type="bibr" rid="ref8">8</xref>], we argue that vocabulary segmentation alone can influence the susceptibility of models to gradient-inversion attacks. Tokenizers trained on domain-specific corpora (eg, RadBERT) are more likely to encode medical terms as single tokens, increasing their semantic coherence; we hypothesized that this would also make such terms easier to reconstruct&#x2014;a prediction our controlled experiments ultimately did not support (see the &#x201C;Results&#x201D; section). In contrast, general-purpose tokenizers tend to fragment clinical phrases into multiple subwords, which may reduce both interpretability and recoverability. This design choice not only impacts downstream clinical utility [<xref ref-type="bibr" rid="ref22">22</xref>] but might, we hypothesized, also alter the reidentifiability of sensitive entities during model inversion&#x2014;a prediction our controlled experiments did not support. By decoupling the tokenizer from the model architecture, our approach provides a targeted evaluation of how vocabulary structure relates to the balance between semantic fidelity and patient privacy.</p><p>In our controlled evaluation, the radiology-specific RadBERT [<xref ref-type="bibr" rid="ref24">24</xref>] tokenizer did not yield significantly higher reconstruction fidelity under attack than general-purpose tokenizers such as GPT-2 [<xref ref-type="bibr" rid="ref25">25</xref>] and LLaMA-2 [<xref ref-type="bibr" rid="ref26">26</xref>]. This indicates that the advantages of domain-adapted pretraining for capturing clinical semantics do not come at the cost of measurably greater gradient-inversion leakage; the vulnerability is intrinsic to the attack and is governed primarily by batch size&#x2014;though under this imprint-probe construction the batch-size effect operates proximately through increased bin-collision burden rather than as a general law of federated training. This observation aligns with evidence that tokenizer and model design choices can influence performance in domain-specific applications, raising the question of whether they likewise affect privacy risk under gradient inversion. Building on this insight, we investigate the extent to which radiology report data can be reconstructed from transformer-based models trained in a simulated FL environment. By orchestrating targeted attacks in a multiclient setup using publicly available radiological datasets, we expose critical vulnerabilities in current FL frameworks. Our study deliberately focuses on the attack surface and does not evaluate defense mechanisms or privacy-preserving strategies. Rather, it aims to characterize worst-case risks and motivate future work toward robust, domain-aware privacy protections for safe deployment of clinical foundation models.</p></sec><sec id="s1-2"><title>Scope and Contributions</title><p>This work establishes a foundational analysis of the privacy vulnerabilities inherent in applying transformer-based FL to radiology reports. We focus on characterizing the attack surface by quantifying the reconstruction risk under a worst-case gradient-inversion scenario. To this end, we provide a systematic comparison of information leakage across major tokenizers, namely, GPT-2, RadBERT, and LLaMA-2. Our central contribution is a controlled demonstration that, with model architecture held fixed, tokenizer choice&#x2014;including domain-specific tokenizers like RadBERT&#x2014;does not significantly change the risk of leaking sensitive patient data, whereas batch size does; practitioners therefore should not, in this setting, rely on tokenizer selection as a privacy safeguard. These findings provide a necessary benchmark and guidepost for developing effective privacy-preserving defenses in medical FL systems. Clinical utility is not directly evaluated in this study and is inferred from prior work establishing improved downstream performance with domain-specific clinical tokenizers.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Ethical Considerations</title><p>This study is a secondary computational analysis of 2 publicly available, previously deidentified clinical text corpora: the dischargesum dataset [<xref ref-type="bibr" rid="ref27">27</xref>] and the MIMIC-CXR (Medical Information Mart for Intensive Care Chest X-Ray) free-text radiology reports accessed via PhysioNet [<xref ref-type="bibr" rid="ref28">28</xref>] under the standard PhysioNet credentialed-access Data Use Agreement. No new data from human participants were collected as part of this work. No identifiable individuals appear in any figure, table, or supplementary material. No participants were recruited, and no compensation was provided. Per institutional policy at the German Cancer Research Center (DKFZ) and consistent with PhysioNet&#x2019;s terms of use, secondary computational analysis of these deidentified public corpora qualifies for ethics board exemption; the original institutional review board approvals covering the primary collection of MIMIC-CXR (Beth Israel Deaconess Medical Center) and Dischargesum extend to secondary research use without additional consent. All experiments were performed on local DKFZ compute infrastructure and no patient data were transmitted to external services.</p></sec><sec id="s2-2"><title>FL and Transformer Models</title><p>FL [<xref ref-type="bibr" rid="ref11">11</xref>] is a decentralized machine learning paradigm that allows multiple institutions to collaboratively train a common model without exchanging raw data. Each client trains locally and sends only gradient updates to a central server for aggregation. In our implementation, we simulated this multi-institutional setup using a GPT-2&#x2013;style transformer (instantiated as a compact 3-layer encoder with token-embedding dimension m=96, 8 attention heads, feed-forward hidden size 1536, and approximately 13.4 M trainable parameters, following the field guide to FL baseline architecture [<xref ref-type="bibr" rid="ref29">29</xref>]) combined with 3 different tokenizers. The overall attack pathway is illustrated in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><list list-type="bullet"><list-item><p>GPT-2 tokenizer: 50,257 tokens</p></list-item><list-item><p>StanfordAIMI/RadBERT tokenizer: 30,522 tokens</p></list-item><list-item><p>LLaMA-2-7B tokenizer: 32,000 tokens</p></list-item></list><p>All tokenizers operated with a fixed sequence length of 32 tokens, so any observed privacy differences are attributable to vocabulary segmentation rather than to input shape or representational scale. Holding the foundation-model architecture fixed at this small, well-characterized baseline allows the tokenizer to serve as the sole independent variable across our GPT-2, RadBERT, and LLaMA-2 comparisons. The 32-token sequence length follows the field guide to FL baseline configuration [<xref ref-type="bibr" rid="ref29">29</xref>] and is consistent with prior gradient-inversion experiments on text models [<xref ref-type="bibr" rid="ref30">30</xref>]; it provides a controlled, computationally tractable regime in which the imprint construction&#x2019;s bin disentanglement is well-characterized. To emulate real multi-institutional training, we randomly partitioned 2 radiology text datasets across 6 cloud&#x2010;based clients: the public Dischargesum corpus (98,206 discharge summaries; 368,751 diagnostic reports) and a locally curated sample of 1500 free-text reports from MIMIC-CXR. Training was repeated with batch sizes of 64, 128, and 256 sequences to measure how aggregation granularity affects privacy leakage in shared gradients. Batch sizes refer throughout to 32-token sequences, not whole reports: each report is tokenized and split into multiple nonoverlapping 32-token windows; a single discharge summary or radiology report typically contributes between 5 and 40 sequences depending on its length, so the per-client local datasets contain &#x2265;5000 sequences and comfortably exceed the largest batch size of 256. Gradient-inversion attacks of the type considered here operate on the gradient produced by a single client during a single training round, which constitutes the standard threat-surface unit in the gradient-leakage literature [<xref ref-type="bibr" rid="ref30">30</xref>]; privacy leakage was therefore quantified per (client and round) rather than over the aggregated federation-level update.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Gradient-inversion attack pathway in federated radiology natural language processing. A malicious central server coordinates model training across multiple hospitals. By inserting a lightweight linear probe at the embedding stage, the server exposes token-level representations and, from the probe&#x2019;s gradients, reconstructs token embeddings in closed form. Decoding these embeddings via the tokenizer&#x2019;s embedding table yields patient text, illustrating that federated learning can leak confidential report content even without raw data sharing.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e88390_fig01.png"/></fig></sec><sec id="s2-3"><title>Threat Model</title><sec id="s2-3-1"><title>Overview</title><p>We characterize the adversary in this study as an active malicious server, following the worst-case threat-model conventions used in prior FL gradient-inversion analyses [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref32">32</xref>]. Formally, the adversary controls the central FL server and possesses the following capabilities. (C1) Architecture modification (active, white box on the server side): the server may insert auxiliary parameters into the shared model before the first communication round, provided the resulting model still trains end-to-end. We exercise this capability by inserting a single analytic imprint module (see section &#x201C;Background and Preliminaries&#x201D;) immediately before the positional embedding. (C2) Per-round per-client gradient observation: the server observes the full set of weight and bias gradients &#x2207;&#x03B8; &#x2112; returned by each client at each round. We do not assume access to intermediate activations, hidden states, or any nongradient signal&#x2014;that is, the protocol is standard FedAvg/FedSGD, not split learning. (C3) Knowledge of the public model and tokenizer: the server knows the model architecture (since it distributed it) and the tokenizer&#x2019;s embedding table (since it is part of the shared model). The adversary does not observe raw client data, client-side optimization state, or any cryptographic key material. We assume no defenses are enabled at any layer of the FL stack: no differential privacy (DP), no secure aggregation, no client-side architecture verification, no gradient anomaly detection. This intentionally maximal threat model isolates the upper-bound leakage attributable to gradient inversion alone, which is the quantity our experiments seek to characterize.</p></sec><sec id="s2-3-2"><title>Detectability and Stealthiness</title><p>The architecture modification in C1 is in principle detectable by any client that compares the received model&#x2019;s parameter signature to a published reference. In current production FL frameworks, no such signature verification is mandatory, and small architectural modifications could in principle be disguised by representing the imprint module as a benign feature-extractor or input-normalization layer. However, the imprint module itself adds approximately (1000&#x00D7;96) + (96&#x00D7;1000)+1000&#x2248;193 K parameters&#x2014;roughly 1.4% of the 13.4-million-parameter base model&#x2014;and would therefore be detectable by any routine client-side parameter-size audit or cryptographic attestation of the published model signature. We therefore treat detectability as a deployment-policy mitigation, not a fundamental defense, and we recommend in section &#x201C;Practical and Regulatory Implications&#x201D; that mandatory client-side architecture attestation be considered as part of clinical FL audit protocols.</p></sec><sec id="s2-3-3"><title>Scope of These Findings</title><p>The reported reconstruction rates correspond to a worst-case adversary: a malicious central server that (1) owns and modifies the shared model architecture before distributing it to clients, (2) observes per-round per-client gradients on weight and bias parameters, and (3) operates in a setting with no privacy defenses enabled&#x2014;no DP, no secure aggregation, no gradient clipping, and no client-side model integrity verification. Under typical real-world cross-silo FL deployments, where institutional clients verify the received model architecture, secure aggregation conceals individual client updates, and lightweight differential-privacy noise is applied, these reconstruction rates would be expected to fall substantially. We therefore interpret our numbers as an upper bound on residual leakage when defenses are absent, not as expected leakage in defended deployments. This framing is consistent with the worst-case methodology of prior gradient-inversion studies [<xref ref-type="bibr" rid="ref30">30</xref>].</p></sec></sec><sec id="s2-4"><title>Background and Preliminaries</title><p>Let the token-embedding dimension be m, and the probe output dimension be k. For a single token position, the hidden vector x &#x2208; &#x211D;<sup>m</sup> is produced by the embedding table before positional addition and attention mixing. We insert a small linear probe (an &#x201C;imprint&#x201D; module in the sense of [<xref ref-type="bibr" rid="ref30">30</xref>]) on top of this hidden state:</p><p><inline-formula><mml:math id="ieqn1"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>&#x22C5;</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mi>x</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>b</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> with W&#x2080; &#x2208; &#x211D;<sup>k&#x00D7;m</sup>, b&#x2080; &#x2208; &#x211D;<sup>k</sup>, and W&#x2082; &#x2208; &#x211D;<sup>m&#x00D7;k</sup>, where k denotes the number of probe bins. The probe is appended once at initialization and trained jointly with the rest of the model. As in standard cross-silo FL (FedAvg [<xref ref-type="bibr" rid="ref11">11</xref>]), each client transmits only the parameter gradient &#x2207;&#x03B8; &#x2112; of its local loss back to the server; no intermediate activations or hidden states are exchanged. This rules out split-learning style leakage and isolates the threat surface to the gradient update itself.</p><p>We initialize b&#x2080; to the cumulative quantiles of a Laplace prior over the expected preactivation distribution and W&#x2080; to a random Gaussian matrix; W&#x2082; is initialized so that <italic>y</italic>&#x2248;<italic>x</italic> in expectation, leaving downstream training behavior essentially unchanged. Under this construction, the bins partition the input space into cumulative half-spaces of geometrically decreasing measure, so that for sufficiently large k at most one sample in a mini-batch activates each bin after consecutive-bin differencing. In general, batch aggregation destroys per-sample information: the transmitted gradient &#x2207;&#x03B8; &#x2112; averages over all B samples in the batch and is not invertible without optimization [<xref ref-type="bibr" rid="ref30">30</xref>]. The cumulative-bin imprint construction circumvents this barrier by analytic design rather than by optimization&#x2014;the contribution of each individual token to &#x2207;W&#x2080; and &#x2207;b&#x2080; is structurally separated by bin index i, so the server can isolate per-sample gradients even though it only ever observes the batch sum.</p><p>From these 2 captured gradients, the input hidden vector x can be recovered in closed form without optimization: <inline-formula><mml:math id="ieqn2"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x25BD;</mml:mi><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>&#x25BD;</mml:mi><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula></p><p>where i is the bin uniquely activated by token j after the cumulative-bin differencing step &#x2207;W<sub>&#x2080;,i</sub> &#x2190; &#x2207;W<sub>&#x2080;,i</sub> &#x2212; &#x2207;W<sub>&#x2080;,i&#x2212;1</sub> (and analogously for &#x2207;b<sub>&#x2080;,i</sub>). We apply this per token and per sample in a batch; sentence reconstructions are formed by decoding each <inline-formula><mml:math id="ieqn3"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mtext> </mml:mtext></mml:math></inline-formula>to the nearest token in the active tokenizer&#x2019;s embedding table E &#x2208; &#x211D;<sup>V&#x00D7;m</sup> under cosine similarity, where V is the tokenizer vocabulary size. We set k=1000 bins and m=96, with <inline-formula><mml:math id="ieqn4"><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mtext> </mml:mtext></mml:math></inline-formula>L2-normalized before decoding; when &#x2016;&#x2207;b&#x2080;,i&#x2016;&#x2082; is below a numerical threshold &#x03B5;, we add a ridge term &#x03B5; to the denominator for numerical stability.</p><p>Ties or near-ties (cosine margin &#x003C; &#x03B4;) among candidate tokens are broken by rescoring with the model&#x2019;s language-model-head logits in a server-side forward pass conditioned on the previously-decoded prefix of the same sample&#x2014;feasible because the malicious server holds the full shared model and can run forward passes on partial reconstructions. Token j is therefore resolved in left-to-right context rather than in isolation; the very first position falls back to the unconditional logit prior.</p></sec><sec id="s2-5"><title>Attack Implementation</title><sec id="s2-5-1"><title>Overview</title><p>We implement the gradient-inversion attack as a deterministic, server-side procedure operating on per-round gradients. The analytic probe and closed-form inversion are defined in the section &#x201C;Background and Preliminaries&#x201D;; below we describe the practical instrumentation, data capture, and decoding steps used in our experiments.</p></sec><sec id="s2-5-2"><title>Step 1: Instrumentation</title><p>The server augments the shared transformer with a lightweight linear probe <inline-formula><mml:math id="ieqn5"><mml:mi>y</mml:mi><mml:mtext>=</mml:mtext><mml:mi>W</mml:mi><mml:mi>x</mml:mi><mml:mtext>+</mml:mtext><mml:mi>b</mml:mi></mml:math></inline-formula> inserted immediately after the token embedding lookup and before positional additions and attention mixing (see the &#x201C;Background and Preliminaries&#x201D; section). This placement exposes token-level representations prior to any token mixing, ensuring reconstructed vectors reflect tokenizer segmentation rather than later architectural effects.</p></sec><sec id="s2-5-3"><title>Step 2: Gradient Capture</title><p>During each client training round, the server records the probe gradients returned in the aggregation step&#x2014;specifically, the per-round weight gradient &#x2207;W&#x2080; &#x2112; &#x2208; &#x211D;<sup>k&#x00D7;m</sup> and bias gradient &#x2207;b&#x2080; &#x2112; &#x2208; &#x211D;<sup>k</sup> of the imprint module&#x2019;s first linear layer (notation as in &#x201C;Background and Preliminaries&#x201D;). The &#x201C;upstream signal&#x201D; &#x2202;&#x2112;/&#x2202;y referenced in prior gradient-inversion literature is not transmitted by clients; it is reconstructed server-side from the captured weight and bias gradients under the cumulative-bin construction. These per-round gradients are stored for subsequent inversion and auditing; we retain round identifiers and batch metadata to link reconstructions to training conditions (dataset, tokenizer, and batch size).</p></sec><sec id="s2-5-4"><title>Step 3: Closed-Form Inversion and Stabilization</title><p>For each captured token gradient in a round, we compute the closed-form embedding estimate <inline-formula><mml:math id="ieqn6"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> following the expression in the &#x201C;Background and Preliminaries&#x201D; section. After inversion, each <inline-formula><mml:math id="ieqn7"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> is L2-normalized to match embedding norm scaling; when &#x2016;&#x2207;b&#x2080;,i&#x2016;&#x2082; is near zero, we add a small ridge term &#x03B5; to the denominator for numerical stability (implementation: = <inline-formula><mml:math id="ieqn8"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>&#x25BD;</mml:mi><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x25BD;</mml:mi><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mi>&#x03B5;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mtext>&#x00A0;</mml:mtext><mml:mi>&#x03F5;</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mn>8</mml:mn></mml:mrow></mml:msup></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>.</p></sec><sec id="s2-5-5"><title>Step 4: Token Decoding</title><sec id="s2-5-5-1"><title>Overview</title><p>Each recovered vector <inline-formula><mml:math id="ieqn9"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>  is decoded by nearest-neighbor search over the active tokenizer&#x2019;s embedding table using cosine similarity (top-1). When multiple vocabulary entries are within a small cosine-similarity margin &#x03B4; of the maximum, the server breaks the tie by rescoring candidates using the model&#x2019;s language-model-head logit at that position, computed in a server-side forward pass conditioned on the previously-decoded prefix. This is feasible because the malicious server holds the full shared model (it distributed the model in the first place) and can execute forward passes on partial reconstructions; the very first position falls back to the unconditional logit prior. Decoded tokens are concatenated in positional order to yield sentence-level reconstructions.</p></sec><sec id="s2-5-5-2"><title>Clarification on Positional Order Recovery</title><p>Positional order is determined by an optimal-assignment matching between recovered breached embeddings and the 32 known positional embeddings of the sequence, implemented as scipy.optimize.linear_sum_assignment (the Hungarian algorithm) in the breaching library, following Fowl et al [<xref ref-type="bibr" rid="ref30">30</xref>]. The language-model-head logit rescoring described above operates within each assigned position to resolve vocabulary ambiguity; it does not itself determine position. The phrase &#x201C;immediately before the positional embedding&#x201D; in our previous revision refers to topological placement within the encoder stack (before the attention and token-mixing layers); the recovered breached embeddings nonetheless carry the positional signature used by the matching stage above.</p></sec></sec></sec><sec id="s2-6"><title>Step 5: Design for Fair Tokenizer Comparison</title><sec id="s2-6-1"><title>Overview</title><p>To isolate tokenization effects from architecture or training hyperparameters, we hold the model architecture, embedding dimension m, probe output dimension <italic>k</italic>, and probe placement constant across experiments. The same inversion and decoding hyperparameters (ridge &#x03B5;, L2 normalization, and nearest-neighbor metric) were applied to GPT-2, RadBERT, and LLaMA-2 tokenizer evaluations without per-tokenizer tuning. Because decoding uses each tokenizer&#x2019;s own embedding table, reconstruction fidelity reflects vocabulary segmentation and tokenizer-induced exposure (single-token entities vs fragmented subwords).</p></sec><sec id="s2-6-2"><title>Implementation Notes and Novelty</title><p>All inversion and decoding steps were implemented in Python (NumPy/PyTorch). Nearest-neighbor lookup used a brute-force cosine search for reproducibility; for large vocabularies, the same pipeline is compatible with approximate nearest-neighbor libraries (Facebook AI Similarity Search [FAISS]). Per-round artifacts (timestamps and batch composition) were retained to permit round-level analyses and the box plots reported in <xref ref-type="fig" rid="figure2">Figure 2</xref> and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref> and <xref ref-type="supplementary-material" rid="app2">2</xref>. The analytic probe itself follows prior work [<xref ref-type="bibr" rid="ref30">30</xref>]; our novelty lies in (1) its embedding-stack placement to isolate tokenizer-specific effects, (2) a tokenizer-controlled, architecture-fixed evaluation showing that tokenizer choice does not significantly change leakage, and (3) clinical entity-level leakage quantification using MedGemma named-entity recognition (NER) [<xref ref-type="bibr" rid="ref33">33</xref>], which quantifies the recovery of clinical concepts (diagnoses, procedures, anatomy, and medications).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Box plots (with the individual runs overlaid as points) showing the distribution of sentence-level bilingual evaluation understudy scores across 5 runs for each tokenizer (GPT-2, RadBERT, and LLaMA-2) on 3 datasets: discharge, diagnosis, and MIMIC-CXR (Medical Information Mart for Intensive Care Chest X-Ray). Each box represents performance at a given batch size (64, 128, and 256), with color-coded distinctions shown in the legend. The plots illustrate both central tendency and variability, showing that the tokenizer distributions overlap substantially, with no consistent tokenizer advantage in reconstruction fidelity. Distributions compress as batch size increases, reflecting reduced sentence-level recoverability. S-BLEU: sentence-level bilingual evaluation understudy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e88390_fig02.png"/></fig></sec><sec id="s2-6-3"><title>Clarification on the Cumulative-Bin Regime</title><p>The k=1000 bins are allocated globally over the batch rather than per token position, so the expected per-bin occupancy is (B&#x00D7;T)/k over the full pool of B&#x00D7;T token activations (<italic>t</italic>=32 here). The strict single-occupancy regime (B&#x003C;k) is therefore not strictly achieved even at our smallest batch: at B=64 the 2048 token activations already exceed the 1000 bins, so a minority of bins are multiply occupied, and multioccupancy grows with batch size until it is pervasive at B=256 (8192 activations). The analytic recovery rule x&#x0302;<sub>j</sub> = &#x2207;W<sub>i</sub>/&#x2207;b<sub>i</sub> is exact only in the single-occupancy regime; under multioccupancy it yields a per-bin weighted average of the active samples&#x2019; embeddings, producing a blurred but nonvacuous reconstruction. Because the recovered vector approximates the occupancy-weighted mean of the colliding token embeddings, its nearest neighbor in the embedding table is generally a token near that centroid rather than any single constituent; empirically, this biases decoding toward high-frequency, generic tokens (common subwords and scaffolding words), so specific low-frequency clinical terms are the first to be lost while frequent structural tokens are still recovered&#x2014;which is why reconstruction degrades gracefully with batch size rather than collapsing to noise. Positional disambiguation in this regime is performed by the optimal-assignment matching stage described in the &#x201C;Attack Implementation&#x201D; section (Step 4) operating on the recovered continuous embeddings. The reconstruction-accuracy degradation we report across batch sizes 64 &#x2192; 128 &#x2192; 256 is consistent with this graceful degradation rather than a strict pigeonhole failure.</p></sec></sec><sec id="s2-7"><title>Dataset and Experimental Setup</title><sec id="s2-7-1"><title>Overview</title><p>We evaluated privacy leakage using 3 clinical-text datasets drawn from 2 public corpora. The Dischargesum [<xref ref-type="bibr" rid="ref27">27</xref>] dataset contained 98,206 discharge summaries and 368,751 diagnostic reports, which were evenly distributed across 6 simulated institutional clients. To complement these structured reports, we also included a subset of 1500 free-text MIMIC-CXR [<xref ref-type="bibr" rid="ref28">28</xref>] radiology reports, similarly partitioned across the 6 clients.</p></sec><sec id="s2-7-2"><title>MIMIC-CXR Sample</title><p>From the full MIMIC-CXR free-text corpus (&#x2248;227,000 reports), we sampled 1500 reports uniformly at random without replacement, conditional on each report being nonempty after standard whitespace stripping. No filtering was applied by clinical content, report length, institutional source, or temporal range. The sample identifier list is included in our reproducibility package and available from the corresponding author on reasonable request, subject to the PhysioNet Data Use Agreement.</p><p>Each client trained a GPT-2&#x2013;style transformer locally and transmitted only gradient updates to the central server for aggregation, emulating a standard cross-institutional FL workflow. Reports were uniformly sampled at random across the 6 clients, producing an approximately independent and identically distributed (IID) partition for the purposes of this study. Experiments systematically varied the local batch size (64, 128, and 256 sequences) to evaluate its effect on gradient-level privacy leakage. Every configuration was repeated across 5 independent runs to account for stochastic training variability and to ensure reproducibility; 95% CIs across runs are reported in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref> (eg, 64.7%, 95% CI 60.2&#x2010;69.2, for GPT-2 on discharge at batch size 64), and a 3-way variance decomposition attributed 94% of the variance in exact-sentence accuracy to batch size versus under 1% to tokenizer; exact paired <italic>t</italic> tests are reported in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. Training details and reproducibility configuration are reported in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>. At each measurement point, the malicious server captures the gradient transmitted by one client in a given round and applies the analytic reconstruction described in &#x201C;Background and Preliminaries&#x201D; to that single update, in line with the standard single-step formulation adopted in the gradient-inversion literature [<xref ref-type="bibr" rid="ref30">30</xref>]. Each batch-size configuration was evaluated under this per-client, per-round setting across the 5 independent runs.</p></sec></sec><sec id="s2-8"><title>Evaluation Metrics</title><p>We assessed reconstruction fidelity using 3 complementary metrics chosen to capture both exact and partial text recovery, following established evaluation practice for reconstructed clinical text [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref24">24</xref>]. These metrics together quantify the extent to which patient-identifiable information can be recreated from shared model updates.</p><list list-type="bullet"><list-item><p>Exact sentence accuracy: The percentage of reconstructed 32-token sequence windows&#x2014;the fixed-length units into which reports are tokenized (see the &#x201C;Threat Model&#x201D; section), for which &#x201C;sentence&#x201D; is used as shorthand&#x2014;that match the ground-truth window exactly, reflecting the proportion of complete exposures of patient text. A reconstructed sentence is counted as an exact match if, after the following normalization steps, every token at every position equals the corresponding token in the original sentence: (1) unicode NFKC normalization, (2) lower-casing, (3) collapsing all runs of internal whitespace to a single space, and (4) preserving punctuation as part of the token sequence rather than stripping it. Sentences are extracted from the concatenated stream of decoded 32-token windows by segmenting at terminal punctuation; the metric is therefore evaluated on these reconstructed sentence units rather than directly on the raw 32-token chunks. We report a stricter case-sensitive variant on the same 5-run experiment for completeness; the qualitative ordering of tokenizers is unchanged across both variants. Exact-sentence accuracy is reported as the conservative upper bound on full-sentence exposure: token-level partial recovery (a single recovered diagnostic noun phrase, for instance) can constitute clinically meaningful leakage even when the full sentence is not reconstructed exactly; this regime is captured by the n-gram and longest-common-subsequence metrics described below. The two metric families together bracket the true leakage rate.</p></list-item><list-item><p>Sentence-level bilingual evaluation understudy (S-BLEU) [<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]: Calculates n-gram precision, where an n-gram is a contiguous sequence of n words (eg, unigrams, bigrams, trigrams, and 4-grams). Bilingual evaluation understudy (BLEU) compares 1&#x2010;4 word sequences in the reconstructed sentence against the reference, with a brevity penalty to discourage overly short outputs. Higher scores indicate stronger local overlap and preservation of clinical phrasing.</p></list-item><list-item><p>Recall-oriented understudy for gisting evaluation (ROUGE-L) [<xref ref-type="bibr" rid="ref36">36</xref>]: Measures recall based on the longest common subsequence (LCS) between reconstruction and reference, capturing the extent to which key clinical terms and their order are retained in longer spans.</p></list-item></list><p>By combining these 3 metrics, we quantify both the frequency of perfect reconstructions and the degree of partial text recovery, offering a comprehensive assessment of patient data exposure under our FL attack.</p></sec><sec id="s2-9"><title>Choice of Metrics</title><p>BLEU and ROUGE-L were originally proposed for translation and summarization quality, where their numerical values are bounded by human-level n-gram overlap. In the gradient-inversion context, we use them as information-recovery metrics rather than as quality metrics: a higher BLEU or ROUGE-L between a reconstructed sentence and the original indicates more shared n-gram content&#x2014;that is, a greater amount of original wording recovered. Exact-sentence accuracy provides the strict-recovery upper bound; S-BLEU and ROUGE-L provide the partial-recovery signal that captures cases where most of a sentence is recovered but a small number of tokens are misdecoded. This complementary use of strict and partial metrics is consistent with prior gradient-inversion text studies [<xref ref-type="bibr" rid="ref37">37</xref>]. We deliberately do not adopt semantic-similarity metrics such as BERTScore in the main results because they can credit paraphrastic matches that do not constitute literal information leakage, which is the privacy question of interest here.</p></sec><sec id="s2-10"><title>Named-Entity Reference-Vocabulary Overlap Evaluation</title><sec id="s2-10-1"><title>Overview</title><p>We quantified recovery of clinically meaningful content using the Google MedGemma medical NER model [<xref ref-type="bibr" rid="ref33">33</xref>]. MedGemma was run on the original reports in each dataset to build a per-dataset reference vocabulary of unique clinical entity surface forms (diagnoses, procedures, anatomical structures, and medications). The same model was applied to the reconstructed text for each tokenizer configuration and dataset.</p></sec><sec id="s2-10-2"><title>Normalization and Matching</title><p>Entity strings were lowercased and whitespace-trimmed; internal punctuation was preserved. An entity was counted as recovered if its surface form exactly matched any term in that dataset&#x2019;s reference vocabulary (case-insensitive). Because this is a corpus-level criterion, it measures reference-vocabulary overlap and represents an upper bound on identifiable leakage: it confirms that a clinical term was reconstructed within the batch but does not by itself establish alignment to a specific source report or patient. No fuzzy matching, synonym expansion, or manual adjudication was used.</p></sec><sec id="s2-10-3"><title>Metric</title><p>For each tokenizer and dataset, we report reference-vocabulary overlap (%), defined as the percentage of the dataset&#x2019;s MedGemma-extracted reference clinical-concept vocabulary that appears in the reconstructed text, averaged over 5 random seeds (mean and SD); this is a corpus-level recall measure and is not aligned to the specific source report. We report per-dataset and pooled overlap across datasets.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Quantitative Reconstruction Success</title><p>On the discharge dataset, exact sentence reconstruction accuracy across tokenizers and batch sizes was as follows:</p><list list-type="bullet"><list-item><p>GPT-2 tokenizer: 64.7%/46.4%/27.3% at batch sizes 64/128/256</p></list-item><list-item><p>RadBERT tokenizer: 70%/50.8%/28.5%</p></list-item><list-item><p>LLaMA-2 tokenizer: 67.5%/48.4%/ 27.5%</p></list-item></list><p>On MIMIC-CXR, reconstruction ranged from 28.7% to 74.7% of sentences (<xref ref-type="fig" rid="figure3">Figure 3</xref>), showing that neither tokenizer choice nor larger batch sizes fully mitigates data leakage. Detailed accuracy results are provided in <xref ref-type="supplementary-material" rid="app6">Multimedia Appendix 6</xref>. All reported reconstruction metrics represent the mean of 5 independent runs; 95% CIs across runs are reported in <xref ref-type="supplementary-material" rid="app3">Multimedia Appendix 3</xref>; exact paired <italic>t</italic> tests are reported in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>. Across configurations, per-cell SDs ranged from approximately 1.5 to 7 percentage points.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Reconstruction fidelity across datasets and tokenizers at varying batch sizes. Top row: exact sentence reconstruction accuracy; middle row: sentence-level bilingual evaluation understudy scores; bottom row: ROUGE-L scores. Each column corresponds to a dataset (discharge, diagnosis, and MIMIC-CXR [Medical Information Mart for Intensive Care Chest X-Ray]), and colors indicate tokenizers (GPT-2, RadBERT, and LLaMA-2). Across metrics and datasets, reconstruction fidelity is comparable among the 3 tokenizers, with differences within seed-to-seed variability. Increasing the batch size from 64 to 256 reduces reconstruction quality across all models, illustrating a trade-off between training granularity and patient data privacy. S-BLEU: sentence-level bilingual evaluation understudy.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e88390_fig03.png"/></fig></sec><sec id="s3-2"><title>Impact of Batch Size on S-BLEU</title><p>Increasing the batch size led to a clear reduction in average reconstruction fidelity across all metrics (<xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>; <xref ref-type="fig" rid="figure2">Figure 2</xref>). For discharge reports, mean S-BLEU decreased from 0.69 at a batch size of 64 to 0.31 at 256; diagnosis reports fell from 0.68 to 0.31; and MIMIC-CXR from 0.71 to 0.32. Parallel trends were observed for ROUGE-L, confirming that coarser gradient aggregation generally mitigates privacy leakage. Notably, across datasets and batch sizes, the 3 tokenizers achieved statistically indistinguishable reconstruction scores. This indicates that domain-specific tokenization does not measurably facilitate more accurate recovery of clinical language, and that tokenizer choice is not a privacy-relevant design lever in this setting. Corresponding mean S-BLEU and ROUGE-L scores are summarized in <xref ref-type="supplementary-material" rid="app7">Multimedia Appendix 7</xref>.</p></sec><sec id="s3-3"><title>Variability in Gradient-Inversion Severity Across Training Rounds</title><p>While average reconstruction fidelity declined with larger batch sizes across all metrics, individual training rounds revealed substantial fluctuations in leakage severity. We illustrate this variability using S-BLEU in <xref ref-type="fig" rid="figure2">Figure 2</xref>, as it provides a balanced measure of local n-gram precision and sentence-level structure, making it particularly sensitive to round-to-round differences. However, similar variability patterns were also observed for exact sentence reconstruction accuracy and ROUGE-L, which are summarized in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendices 1</xref> and <xref ref-type="supplementary-material" rid="app2">2</xref>. Even at the highest batch size (256), certain rounds produced high-fidelity reconstructions comparable to those at batch size 64, underscoring a persistent privacy risk. Coarser gradient aggregation therefore reduces but does not eliminate patient information exposure in FL.</p><p>In real-world medical AI deployments, even a single outlier round with high leakage could compromise patient confidentiality. Therefore, it is essential to characterize not just average-case leakage but also variability across rounds, to ensure that system-level guarantees account for worst-case scenarios.</p></sec><sec id="s3-4"><title>Clinical-Concept Reference-Vocabulary Overlap in Reconstructed Text</title><p>As presented in <xref ref-type="table" rid="table1">Table 1</xref>, reference-vocabulary overlap was high and statistically indistinguishable across tokenizers: MedGemma-extracted clinical-concept overlap averaged 75.1% (GPT-2), 74.1% (RadBERT), and 72.9% (LLaMA-2) across datasets and seeds, with no pairwise difference reaching significance.</p><p>These findings demonstrate that gradient-inversion attacks can reveal not only structural sentence fragments but also clinically meaningful entities, posing a tangible privacy risk. That overlap remains high (approximately 75%) regardless of the tokenizer indicates that this leakage is intrinsic to the attack rather than a property of any particular tokenizer in federated training settings.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Recovery of clinically meaningful terms from gradient-inverted reconstructions using GPT-2, LLaMA-2, and RadBERT tokenizers<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Tokenizer</td><td align="left" valign="bottom">Discharge, mean (SD)</td><td align="left" valign="bottom">Diagnosis, mean (SD)</td><td align="left" valign="bottom">MIMIC-CXR<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, mean (SD)</td><td align="left" valign="bottom">Pooled, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">GPT-2</td><td align="left" valign="top">71.0 (5.7)</td><td align="left" valign="top">73.0 (3.5)</td><td align="left" valign="top">81.3 (7.1)</td><td align="left" valign="top">75.1 (7.0)</td></tr><tr><td align="left" valign="top">RadBERT</td><td align="left" valign="top">73.7 (5.8)</td><td align="left" valign="top">72.4 (7.9)</td><td align="left" valign="top">76.3 (7.6)</td><td align="left" valign="top">74.1 (6.8)</td></tr><tr><td align="left" valign="top">LLaMA-2</td><td align="left" valign="top">71.8 (7.6)</td><td align="left" valign="top">72.6 (3.0)</td><td align="left" valign="top">74.2 (10.1)</td><td align="left" valign="top">72.9 (7.0)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Clinical named entities were identified via the Google MedGemma model across all 3 datasets and 5 random seeds. Values are clinical-concept reference-vocabulary overlap (%)&#x2014;the percentage of each dataset&#x2019;s reference clinical-concept vocabulary appearing in reconstructions, a corpus-level recall measure&#x2014;mean (SD); differences across tokenizers are not statistically significant. Pooled pairwise comparisons (n=15): RadBERT vs GPT-2 &#x0394;=&#x2212;1.0 pp; RadBERT vs LLaMA-2 <italic>&#x0394;</italic>=+1.2 pp; GPT-2 vs LLaMA-2 <italic>&#x0394;</italic>=+2.2 pp&#x2014;none statistically significant.</p></fn><fn id="table1fn2"><p><sup>b</sup>MIMIC-CXR: Medical Information Mart for Intensive Care Chest X-Ray.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Qualitative Reconstruction</title><p><xref ref-type="table" rid="table2">Tables 2</xref><xref ref-type="table" rid="table3"/>-<xref ref-type="table" rid="table4">4</xref> showcase original and reconstructed outputs from the dischargesum, radiology, and MIMIC-CXR datasets, illustrating the clinical content recoverable across all 3 tokenizers (aggregate metrics show no significant tokenizer difference). These examples use oracle alignment for readability only: the positional ordering shown is not available to the attacker, who observes an unordered or partially ordered set of recovered tokens, so the aligned display illustrates recoverable content rather than an attacker&#x2019;s verbatim output. Across the reconstructions, all 3 tokenizers recovered substantial clinical content while also dropping or fragmenting individual terms; in the specific example shown, recovered entities included terms such as &#x201C;catheter&#x201D; and &#x201C;nodularity.&#x201D; Which terms were retained or dropped varied from example to example, and&#x2014;consistent with the aggregate analysis, which found no significant tokenizer difference&#x2014;these single-example observations should not be read as a systematic tokenizer ordering. These qualitative results reinforce the quantitative findings, demonstrating that even domain-adapted transformer models remain susceptible to gradient-based inversion attacks, recovering not only template phrases but also clinically meaningful patient information.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Discharge report: reconstruction across tokenizers (GPT-2, LLaMA-2, and StanfordAIMI/RadBERT; Oracle-aligned visualization for readability; not attacker-observable output)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Discharge report: reconstruction across tokenizers</td><td align="left" valign="bottom">Sample Report Text</td></tr></thead><tbody><tr><td align="left" valign="top">Original report</td><td align="left" valign="top">Dear Mr. Alex, It Was A Pleasure Taking Care Of You Here At ABC clinic. You Were Admitted To Our Hospital After Undergoing Repair Of Your Ventral Hernia. You Have Recovered From Surgery And Are Now Ready To Be Discharged To Home With Services. Please Follow The Recommendations Below To Ensure A Speedy And Uneventful Recovery. ACTIVITY: - Do not drive until you have stopped taking pain medicine and feel you could respond in an emergency. - You may climb stairs. - You may go outside, but avoid traveling long distances until you see your surgeon at your next visit.</td></tr><tr><td align="left" valign="top">GPT-2</td><td align="left" valign="top">Dear Mr. Alex, It Was A [DROP] Taking Care Of You [DROP] At ABC clinic. You Were Admitted To Our Hospital After Undergoing Repair Of Your Ventral Hernia. You Have Recovered From Surgery And Are Now Ready To Be Discharged To Home With Services. Please Follow The Recommendations [DROP] To Ensure A And [DROP] Recovery. ACTIVITY: - Do [DROP] drive until you have [DROP] taking pain medicine and feel you could respond in an emergency. - You may climb stairs. - You may go outside, but avoid traveling [DROP] distances until you [DROP] your surgeon at your next visit.</td></tr><tr><td align="left" valign="top">LLaMA-2</td><td align="left" valign="top">[DROP] Mr. Alex, It [DROP] A Pleasure Taking Care Of You Here At ABC clinic. You Were Admitted To Our Hospital After Undergoing Repair Of Your Ventral [DROP]. You Have [DROP] From [DROP] And [DROP] [DROP] [DROP] To Be Discharged To Home With Services. Please Follow The [DROP] Below To Ensure A Speedy And Uneventful Recovery. ACTIVITY: - Do not drive until you have stopped taking pain [DROP] and feel you [DROP] [DROP] in an emergency. - [DROP] [DROP] climb stairs. - You may go outside, but avoid traveling long distances until you see your surgeon at your next visit.</td></tr><tr><td align="left" valign="top">StanfordAIMI/RadBERT</td><td align="left" valign="top">Dear Mr. Alex, It Was A [DROP] Taking Care Of You Here At ABC clinic. You Were Admitted To Our Hospital After Undergoing Repair Of Your Ventral Hernia. You Have Recovered From [DROP] And Are Now Ready To Be Discharged To Home With Services. Please Follow The Recommendations Below To Ensure A Speedy And Uneventful Recovery. ACTIVITY: - Do not drive until you have stopped taking pain medicine and feel you could [DROP] in an [DROP]. - You may climb stairs. - You may go outside, but avoid traveling [DROP] [DROP] until you see your [DROP] at your next visit.</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>The original report and reconstructions are aligned by model. Recovered content varies across models in this single example; aggregate reconstruction metrics do not differ significantly across tokenizers (<xref ref-type="fig" rid="figure3">Figure 3</xref>; <xref ref-type="table" rid="table1">Table 1</xref>). The [DROP] marker indicates unreconstructed spans (native source redactions in <xref ref-type="table" rid="table3">Table 3</xref> are preserved as asterisks). Note on alignment. The [DROP] marker indicates token positions for which the gradient-inverted reconstruction did not produce a valid token (a bin collision or numerical-instability bin). For visual clarity, these positions are aligned against the original text using oracle alignment, which the attacker does not have access to in a real attack setting; the attacker would observe an unordered or partially ordered set of recovered tokens without knowledge of where reconstruction failed. Oracle alignment is used here purely for reader interpretability and does not reflect adversary capability.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Radiology report: comparison of reconstructed output across tokenizers; Oracle-aligned visualization for readability; not attacker-observable output)<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Radiology report: reconstruction across tokenizers</td><td align="left" valign="bottom">Sample Report Text</td></tr></thead><tbody><tr><td align="left" valign="top">Original report</td><td align="left" valign="top">EXAMINATION: LIVER OR GALLBLADDER US (SINGLE ORGAN) INDICATION: History: with cirrhosis, increased abdominal pain TECHNIQUE: Gray scale and color Doppler ultrasound images of the right upper quadrant were obtained. COMPARISON: Abdominal ultrasound from **** FINDINGS: The liver is extremely coarse and nodular in echotexture similar to the prior examination consistent with a history of cirrhosis. Parenchymal heterogeneity limits detection of focal lesions.</td></tr><tr><td align="left" valign="top">GPT-2</td><td align="left" valign="top">[DROP]: LIVER OR GALLBLADDER US (SINGLE ORGAN) INDICATION: History: with [DROP], increased abdominal pain TECHNIQUE: Gray scale and color [DROP] ultrasound images of the right upper quadrant were obtained. COMPARISON: Abdominal ultrasound from **** FINDINGS: The liver is [DROP] coarse and [DROP] in echotexture similar to the prior examination consistent with a history of [DROP]. Parenchymal heterogeneity limits detection of focal [DROP].</td></tr><tr><td align="left" valign="top">LLaMA-2</td><td align="left" valign="top">EXAMINATION: [DROP] OR GALLBLADDER US ([DROP] ORGAN) [DROP]: History: with cirrhosis, [DROP] abdominal pain TECHNIQUE: Gray scale and color [DROP] ultrasound images of the right upper quadrant were obtained. COMPARISON: Abdominal ultrasound from **** FINDINGS: The liver is extremely [DROP] and nodular in [DROP] similar to [DROP] prior examination consistent with a history of cirrhosis. Parenchymal [DROP] limits detection of focal lesions.</td></tr><tr><td align="left" valign="top">StanfordAIMI/RadBERT</td><td align="left" valign="top">EXAMINATION: LIVER OR GALLBLADDER US (SINGLE ORGAN) INDICATION: History: with cirrhosis, increased abdominal pain TECHNIQUE: Gray scale and color [DROP] ultrasound images of the right upper quadrant were obtained. [DROP]: Abdominal ultrasound from **** FINDINGS: The liver is extremely [DROP] and nodular in echotexture similar to the prior examination consistent with a history of cirrhosis. [DROP] heterogeneity limits detection of focal lesions.</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>In this single example, RadBERT and GPT-2 preserved more contextual phrasing than LLaMA-2; aggregate metrics show no significant tokenizer difference (<xref ref-type="fig" rid="figure3">Figure 3</xref>; <xref ref-type="table" rid="table1">Table 1</xref>). Sensitive findings such as &#x201D;cirrhosis&#x201D; and &#x201D;nodularity&#x201D; were partially reconstructed in all models. Note on alignment. The [DROP] marker indicates token positions for which the gradient-inverted reconstruction did not produce a valid token (a bin collision or numerical-instability bin); 4-asterisk sequences (****) in the Original report row are preserved as native source deidentification redactions from the discharge or radiology reports. For visual clarity, these positions are aligned against the original text using oracle alignment, which the attacker does not have access to in a real attack setting; the attacker would observe an unordered or partially ordered set of recovered tokens without knowledge of where reconstruction failed. Oracle alignment is used here purely for reader interpretability and does not reflect adversary capability.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>MIMIC-CXR<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> report: reconstruction outputs across tokenizers; Oracle-aligned visualization for readability; not attacker-observable output)<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">MIMIC-CXR report: reconstruction across tokenizers</td><td align="left" valign="bottom">Sample Report Text</td></tr></thead><tbody><tr><td align="left" valign="top">Original report</td><td align="left" valign="top">10439781,55811525, &#x201C;Frontal and lateral views of the chest were obtained. Left-sided Port-A-Catheter is similar in position, terminating at the cavoatrial/right atrial junction. Patient has diffuse increase in interstitial markings bilaterally consistent with patient&#x2019;s underlying history of chronic interstitial lung disease with likely overlying pulmonary edema improved since ___, but similar in appearance as compared to ___. No definite focal consolidation or pleural effusion. Multilevel vertebroplasties are seen along the thoracic spine, similar to prior.,&#x201D;Pulmonary edema superimposed on known lung fibrosis.</td></tr><tr><td align="left" valign="top">GPT-2</td><td align="left" valign="top">10439781,55811525, &#x201C;Frontal and [DROP] views of the chest were obtained. Left-sided Port-A-[DROP] is similar in position, terminating at the cavoatrial/right atrial junction. Patient has diffuse increase in interstitial markings bilaterally consistent with patient&#x2019;s underlying history of chronic [DROP] lung disease [DROP] likely overlying pulmonary edema improved since ___, but similar in [DROP] as compared to ___. No definite focal consolidation or [DROP] effusion. Multilevel vertebroplasties are seen along the thoracic spine, [DROP] to prior.,&#x201D;Pulmonary edema superimposed on known lung fibrosis.</td></tr><tr><td align="left" valign="top">LLaMA-2</td><td align="left" valign="top">10439781,[DROP], &#x201C;Frontal and lateral [DROP] of the chest were [DROP]. Left-sided Port-A-Catheter is similar in position, terminating at the cavoatrial/right atrial junction. Patient has diffuse increase in interstitial markings [DROP] consistent with patient&#x2019;s underlying history of chronic interstitial lung [DROP] with likely [DROP] pulmonary edema improved since ___, but similar in [DROP] as compared to ___. No [DROP] focal consolidation or pleural [DROP]. Multilevel vertebroplasties are seen along the thoracic spine, similar to prior.,&#x201D;Pulmonary edema superimposed on known lung [DROP].</td></tr><tr><td align="left" valign="top">StanfordAIMI/RadBERT</td><td align="left" valign="top">10439781,55811525, &#x201C;Frontal and [DROP] views of the chest were obtained. Left-sided Port-A-Catheter is similar in position, terminating at the cavoatrial/right atrial junction. Patient has diffuse increase in interstitial markings bilaterally consistent with patient&#x2019;s underlying history of chronic interstitial lung disease with likely overlying pulmonary edema improved since ___, but similar in appearance as compared to ___. No [DROP] focal consolidation or pleural effusion. [DROP] vertebroplasties are seen along the thoracic spine, similar to prior.,&#x201D;Pulmonary edema [DROP] on known lung fibrosis.</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>MIMIC-CXR: Medical Information Mart for Intensive Care Chest X-Ray.</p></fn><fn id="table4fn2"><p><sup>b</sup>GPT-2 and RadBERT captured anatomical and pathological keywords such as &#x201C;fibrosis&#x201D; and &#x201C;catheter&#x201D; more faithfully than LLaMA-2. In this single example, structural fidelity was higher in the RadBERT output; aggregate metrics show no significant tokenizer difference (<xref ref-type="fig" rid="figure3">Figure 3</xref>; <xref ref-type="table" rid="table1">Table 1</xref>). Note on alignment. The [DROP] marker indicates token positions for which the gradient-inverted reconstruction did not produce a valid token (a bin collision or numerical-instability bin). For visual clarity, these positions are aligned against the original text using oracle alignment, which the attacker does not have access to in a real attack setting; the attacker would observe an unordered or partially ordered set of recovered tokens without knowledge of where reconstruction failed. Oracle alignment is used here purely for reader interpretability and does not reflect adversary capability.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Results</title><p>This study demonstrates that even with domain-specific tokenization and larger batch sizes, transformer models in an FL setup leak substantial patient information. The active server reconstructed up to approximately 75% of radiology-report sentences. On the Discharge dataset, exact-sentence accuracy was 64.7%/46.4%/27.3% for GPT-2, 70%/50.8%/28.5% for RadBERT, and 67.5%/48.4%/27.5% for LLaMA-2 at batch sizes 64/128/256. S-BLEU similarly declined with batch size (eg, GPT-2 on discharge reports from 0.66 to 0.31), yet even at the largest batch sizes roughly a quarter of sentences were fully recovered. Across batch sizes and datasets, reconstruction fidelity did not differ significantly among the 3 tokenizers (mean exact-sentence accuracy 48.3% for each; no pairwise comparison significant). After Holm correction for the 27 pairwise comparisons, no tokenizer difference remained significant (all adjusted <italic>P</italic>=1.00), and a 3-way variance decomposition attributed 94% of the variance in exact-sentence accuracy to batch size versus less than 1% to tokenizer, corroborating batch size&#x2014;not tokenizer choice&#x2014;as the dominant factor; the pairwise tokenizer effect sizes were negligible (mean differences &#x2264;0.05 percentage points; Cohen <italic>d</italic>&#x2264;0.03). All 3 tokenizers, including the domain-specific RadBERT, leaked substantial clinical content, and none eliminated leakage. These findings confirm that neither increasing batch sizes nor specialized tokenizers alone can ensure patient privacy in FL&#x2010;trained LLMs.</p><p>To further disambiguate templated language from truly clinical leakage, we applied MedGemma NER to the reconstructed text across all 3 datasets. Across models, approximately 75% of reference clinical entities were recovered (<xref ref-type="table" rid="table1">Table 1</xref>). These findings confirm that privacy risks extend beyond boilerplate phrasing to include clinical concepts relevant to diagnosis and care.</p><p>Our findings underscore that tokenizer design is not a decisive factor in privacy leakage under gradient inversion attacks. In a controlled named-entity analysis, clinical-concept reference-vocabulary overlap was statistically indistinguishable across tokenizers, with approximately 75% of each dataset&#x2019;s reference vocabulary appearing under each (GPT-2 75.1%, RadBERT 74.1%, LLaMA-2 72.9%; no pairwise difference significant). This indicates that domain-specific vocabulary segmentation does not, by itself, make reidentification-relevant clinical content easier to reconstruct. Within the evaluated architecture, attack, datasets, and metrics, tokenizer choice should therefore not be treated as a privacy safeguard; batch size and explicit defenses are the privacy-relevant levers. Future privacy audits and risk assessments of federated clinical models should focus on batch size and explicit defenses rather than tokenizer choice, especially as domain-specific LLMs become more prevalent in medical AI.</p></sec><sec id="s4-2"><title>Hypothesized Mechanism and Why It Is Not Borne Out</title><p>We initially reasoned that domain-specific tokenization should heighten reconstruction. The reconstruction step decodes each recovered token-embedding vector back to the nearest token in the active tokenizer&#x2019;s embedding table. A clinical concept such as &#x201C;pneumothorax&#x201D; is encoded as a single token in RadBERT&#x2019;s domain-specific vocabulary, so a successful nearest-neighbor decode would yield the entire concept verbatim, whereas the same concept is segmented into multiple subwords (eg, &#x201C;p/neu/moth/orax&#x201D;) in GPT-2&#x2019;s byte-pair encoding, where recovering it requires every subword embedding to be reconstructed correctly and placed in the correct positional order. Under this reasoning, single-token clinical terms should present fewer points of failure. We tested this prediction directly&#x2014;including a prespecified analysis restricted to clinical terms that GPT-2 fragments into multiple subwords, where any single-token advantage should be largest&#x2014;and found no significant tokenizer difference in exact-sentence accuracy, clinical-concept overlap, or the fragmented-term subset. The hypothesized single-token advantage is therefore not borne out empirically: across the datasets and batch sizes studied, the closed-form attack recovers clinical content at comparable rates regardless of tokenizer, and batch size is the dominant factor. This controlled negative result indicates that, in this setting, tokenizer selection should not be treated as a privacy safeguard.</p></sec><sec id="s4-3"><title>Clinical Concepts Versus Directly Identifiable Protected Health Information</title><p>The MedGemma NER model used in this study captures clinical concepts (diagnoses, procedures, anatomical structures, and medications) rather than direct identifiers under the HIPAA Safe Harbor enumeration (names, medical record numbers, dates more granular than year, full-face photographic images, and so on). Direct identifiers are largely already redacted in the public Dischargesum and MIMIC-CXR corpora before release. The recovered terms therefore reflect reidentification potential through clinical context&#x2014;for example, an unusually rare diagnosis, a unique procedure pattern, or a specific medication regimen that, in combination with auxiliary recovered context, could plausibly contribute to reidentifying a patient&#x2014;rather than direct exposure of an enumerated identifier. We make this distinction explicit because the regulatory framing of &#x201C;PHI&#x201D; is broader than &#x201C;direct identifier&#x201D;; HIPAA&#x2019;s expert-determination standard (45 CFR &#x00A7;164.514(b)(1)) treats clinical context recoverable from a record as part of identifiability risk, and the same logic is reflected in the GDPR&#x2019;s notion of indirect identifiability via singling out.</p></sec><sec id="s4-4"><title>Why Batch Size Reduces Leakage in This Attack</title><p>The imprint module&#x2019;s analytic recovery rule x&#x0302;<sub>j</sub> =&#x2207;W<sub>&#x2080;,i</sub>/&#x2207;b<sub>&#x2080;,i</sub> relies on the assumption that bin i is activated by at most one sample in the batch. With k=1000 bins and batch size B, the expected number of samples per bin is B/k, and the probability that any given bin is hit by &#x2265;2 samples grows approximately quadratically in B (a birthday-paradox argument). When 2 or more samples share a bin, the recovered ratio &#x2207;W<sub>&#x2080;,i</sub>/&#x2207;b<sub>&#x2080;,i</sub> becomes a gradient-weighted average of distinct embeddings rather than a clean single-sample recovery, and the corresponding nearest-neighbor decode is corrupted. This explains why exact-sentence accuracy decays with batch size while never reaching zero: as long as some bins remain singly occupied, the corresponding samples are recovered cleanly. The same argument predicts that increasing k relative to B would partially compensate for batch growth&#x2014;a direction we leave to future work.</p></sec><sec id="s4-5"><title>Key Contribution</title><p>We provide the first systematic evaluation of how different transformer tokenizers impact gradient&#x2010;inversion vulnerability on radiology text, showing that domain&#x2010;specific tokenization (RadBERT) does not significantly change leakage risk relative to general-purpose tokenizers, even as it better preserves clinical terminology, whereas batch size does.</p></sec><sec id="s4-6"><title>Comparison With Prior Work</title><sec id="s4-6-1"><title>Overview</title><p>Our results extend prior demonstrations of gradient-inversion attacks in generic NLP models [<xref ref-type="bibr" rid="ref32">32</xref>,<xref ref-type="bibr" rid="ref38">38</xref>] and imaging domains [<xref ref-type="bibr" rid="ref20">20</xref>] to the structured and template-rich text of radiology reports. Earlier works such as DLG (deep leakage from gradients) [<xref ref-type="bibr" rid="ref38">38</xref>] and iDLG (improved deep leakage from gradients) [<xref ref-type="bibr" rid="ref39">39</xref>] reconstructed generic sentences or pixel-level images from shared gradients but did not examine how tokenizer design mediates privacy leakage. Subsequent privacy-preserving FL frameworks in medical imaging [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref20">20</xref>] primarily focused on architectural defenses rather than language-specific vulnerabilities. Unlike these approaches, our study isolates the tokenizer as a controllable privacy variable by holding the transformer architecture and training hyperparameters constant across GPT-2, RadBERT, and LLaMA-2. This tokenizer-controlled setup reveals that vocabulary segmentation alone does not significantly influence gradient-based reconstruction fidelity. This comparison against prior gradient-inversion literature is summarized in <xref ref-type="table" rid="table5">Table 5</xref>. While Akinci et al [<xref ref-type="bibr" rid="ref21">21</xref>] first discussed text leakage risks in clinical LLMs, no prior work systematically compared multiple tokenization schemes or evaluated reconstruction across both discharge and imaging-report corpora. Our findings therefore provide the first controlled evidence that domain-adapted tokenizers, though beneficial for clinical utility, do not measurably reduce or amplify privacy leakage in federated radiology LLMs; batch size, not tokenizer choice, governs reconstruction risk.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Comparison with prior gradient-inversion studies. To our knowledge, this is the first study to explicitly hold the foundation-model architecture constant and vary only the tokenizer, in the clinical radiology NLP<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup> domain.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Study</td><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Tokenizer controlled?</td><td align="left" valign="bottom">Domain</td><td align="left" valign="bottom">Datasets</td><td align="left" valign="bottom">Defense evaluation?</td></tr></thead><tbody><tr><td align="left" valign="top">DLG<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">LSTM<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup>, ResNet<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">No (single tokenizer per task)</td><td align="left" valign="top">Generic vision + NLP</td><td align="left" valign="top">CIFAR<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup>, MNIST<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup></td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">iDLG<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup></td><td align="left" valign="top">LeNet</td><td align="left" valign="top">No</td><td align="left" valign="top">Generic vision</td><td align="left" valign="top">CIFAR</td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">TAG</td><td align="left" valign="top">BERT<sup><xref ref-type="table-fn" rid="table5fn8">h</xref></sup></td><td align="left" valign="top">Single tokenizer</td><td align="left" valign="top">Generic NLP</td><td align="left" valign="top">CoLA<sup><xref ref-type="table-fn" rid="table5fn9">i</xref></sup>, SST-2<sup><xref ref-type="table-fn" rid="table5fn10">j</xref></sup></td><td align="left" valign="top">None</td></tr><tr><td align="left" valign="top">Robbing the Fed</td><td align="left" valign="top">Transformer/ViT<sup><xref ref-type="table-fn" rid="table5fn11">k</xref></sup></td><td align="left" valign="top">No</td><td align="left" valign="top">Vision + WikiText</td><td align="left" valign="top">WikiText, ImageNet</td><td align="left" valign="top">DP<sup><xref ref-type="table-fn" rid="table5fn12">l</xref></sup> discussed</td></tr><tr><td align="left" valign="top">Hatamizadeh et al [<xref ref-type="bibr" rid="ref20">20</xref>]</td><td align="left" valign="top">UNet, ViT</td><td align="left" valign="top">N/A<sup><xref ref-type="table-fn" rid="table5fn13">m</xref></sup> (vision)</td><td align="left" valign="top">Medical imaging</td><td align="left" valign="top">BraTS<sup><xref ref-type="table-fn" rid="table5fn14">n</xref></sup>, LIDC<sup><xref ref-type="table-fn" rid="table5fn15">o</xref></sup></td><td align="left" valign="top">Discussed</td></tr><tr><td align="left" valign="top">Akinci et al [<xref ref-type="bibr" rid="ref21">21</xref>]</td><td align="left" valign="top">Generic LLMs<sup><xref ref-type="table-fn" rid="table5fn16">p</xref></sup></td><td align="left" valign="top">Discussed, not measured</td><td align="left" valign="top">Clinical NLP</td><td align="left" valign="top">N/A (review article)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn17">q</xref></sup></td></tr><tr><td align="left" valign="top">This work</td><td align="left" valign="top">3-layer Transformer (held constant)</td><td align="left" valign="top">Yes &#x2014; three tokenizers compared</td><td align="left" valign="top">Clinical radiology NLP</td><td align="left" valign="top">Dischargesum, MIMIC-CXR<sup><xref ref-type="table-fn" rid="table5fn18">r</xref></sup></td><td align="left" valign="top">Discussed; full evaluation noted as future work</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>NLP: natural language processing.</p></fn><fn id="table5fn2"><p><sup>b</sup>DLG: deep leakage from gradients.</p></fn><fn id="table5fn3"><p><sup>c</sup>LSTM: long short-term memory.</p></fn><fn id="table5fn4"><p><sup>d</sup>ResNet: residual network.</p></fn><fn id="table5fn5"><p><sup>e</sup>CIFAR: Canadian Institute for Advanced Research.</p></fn><fn id="table5fn6"><p><sup>f</sup>MNIST: Modified National Institute of Standards and Technology database.</p></fn><fn id="table5fn7"><p><sup>g</sup>iDLG: improved deep leakage from gradients.</p></fn><fn id="table5fn8"><p><sup>h</sup>BERT: bidirectional encoder representations from transformers.</p></fn><fn id="table5fn9"><p><sup>i</sup>CoLA: Corpus of Linguistic Acceptability.</p></fn><fn id="table5fn10"><p><sup>j</sup>SST-2: Stanford Sentiment Treebank (binary).</p></fn><fn id="table5fn11"><p><sup>k</sup>ViT: vision transformer.</p></fn><fn id="table5fn12"><p><sup>l</sup>DP: differential privacy.</p></fn><fn id="table5fn13"><p><sup>m</sup>N/A: not applicable.</p></fn><fn id="table5fn14"><p><sup>n</sup>BraTS: Brain Tumor Segmentation (challenge).</p></fn><fn id="table5fn15"><p><sup>o</sup>LIDC: Lung Image Database Consortium.</p></fn><fn id="table5fn16"><p><sup>p</sup>LLM: large language model.</p></fn><fn id="table5fn17"><p><sup>q</sup>Not specified.</p></fn><fn id="table5fn18"><p><sup>r</sup>MIMIC-CXR: Medical Information Mart for Intensive Care &#x2013; Chest X-Ray.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s4-6-2"><title>Practical Significance</title><p>Earlier closed-form gradient inversion attacks on text models report exact-sentence recovery rates of approximately 30%&#x2010;50% at small batch sizes on generic English corpora [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]; our radiology-domain experiments reach comparable and, at small batch sizes, higher rates (27%&#x2010;75%), confirming that the threat extends from generic to clinical NLP and is, if anything, more severe at the strict-recovery level. The practical significance is, however, qualitatively different in the clinical setting: radiology corpora encode clinical context (diagnoses, procedure types, and anatomical findings) at higher density per token than newswire or web text, so equivalent reconstruction rates translate to higher clinical informativeness per recovered sentence. A single recovered sentence in a clinical context may contain a rare diagnosis or unusual procedure combination that contributes meaningfully to reidentification risk, whereas a single recovered sentence from generic web text typically carries no analogous reidentifying signal.</p></sec><sec id="s4-6-3"><title>Scope of the Tokenizer Comparison</title><p>Because we hold the foundation-model architecture fixed at a small, well-characterized baseline, our findings reflect the effect of vocabulary segmentation on gradient-inversion leakage, not the privacy properties of the GPT-2, RadBERT, or LLaMA-2 foundation models themselves. Larger frontier-scale foundation models exhibit different gradient sparsity and noise dynamics that can change inversion vulnerability in either direction; we do not claim that our results extend to those regimes, and explicit characterization at frontier scale is an important target for future work.</p></sec></sec><sec id="s4-7"><title>Practical and Regulatory Implications</title><sec id="s4-7-1"><title>Focus on Attack Characterization</title><p>This study deliberately isolates the gradient&#x2010;inversion attack vector and does not evaluate or compare privacy defenses. Instead, we quantify the worst-case leakage risk under an unconstrained adversary. Subsequent work should build on these findings to empirically evaluate and optimize defense mechanisms in realistic radiology FL pipelines. Our results demonstrate that generic FL defenses such as DP [<xref ref-type="bibr" rid="ref40">40</xref>], secure aggregation [<xref ref-type="bibr" rid="ref41">41</xref>], and gradient&#x2010;anomaly detection have yet to be empirically validated for the structured, high&#x2010;risk text found in radiology reports. Injecting noise via DP may reduce reconstruction fidelity but requires careful calibration to avoid degrading critical diagnostic language. Similarly, cryptographic secure aggregation can obscure individual updates but may introduce prohibitive latency in hospital networks. Real&#x2010;time gradient&#x2010;anomaly detection promises early warning of inversion attacks, yet its thresholds must be tuned to the unique update patterns of clinical text to prevent both false alarms and missed breaches. Because the attack characterized here depends on the server altering the shared model graph (inserting a linear imprint probe), the most direct and low-cost safeguard is client-side model-graph integrity verification: before each local training round, clients should validate the received architecture against an expected specification and verify a cryptographic checksum of the model state and its parameter count, rejecting any model whose computation graph or parameter budget deviates. Production FL frameworks such as NVIDIA FLARE, Intel OpenFL, and Owkin Substra already expose client-side hooks and secure-provisioning mechanisms that can enforce this static computation-graph validation and attestation, making the safeguard practical to deploy today. To ensure compliance with HIPAA and GDPR, health care organizations may benefit from running pilot evaluations of these defenses under realistic conditions and document performance trade&#x2010;offs. Regulatory bodies could consider developing formal audit protocols for FL systems, encompassing threat modeling, reconstruction testing, and defense efficacy, before granting approval for clinical deployment.</p></sec><sec id="s4-7-2"><title>Defenses and Practical Mitigations</title><p>While this study focused on characterizing gradient inversion risk, effective deployment of federated radiology models requires integrating strong privacy defenses [<xref ref-type="bibr" rid="ref41">41</xref>]. Common mitigation strategies such as synthetic data generation [<xref ref-type="bibr" rid="ref42">42</xref>] and DP alone remain insufficient for LLMs [<xref ref-type="bibr" rid="ref43">43</xref>-<xref ref-type="bibr" rid="ref45">45</xref>]. Synthetic data can leak statistical artifacts and fail to preserve downstream clinical performance [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>], whereas DP often requires large privacy budgets to maintain utility, limiting its protective value [<xref ref-type="bibr" rid="ref46">46</xref>].</p><p>In practice, more robust protection arises from hybrid approaches that combine secure aggregation to mask individual updates with task-tuned DP noise, gradient clipping, and auditable privacy monitoring. Cryptographic protocols such as homomorphic encryption or secure multiparty computation offer mathematically proven safeguards but may introduce computational overhead in hospital networks [<xref ref-type="bibr" rid="ref47">47</xref>]. Future implementations of federated clinical NLP should adopt layered defenses that balance efficiency, regulatory compliance, and measurable privacy guarantees [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref47">47</xref>].</p></sec></sec><sec id="s4-8"><title>Limitations</title><p>This study was designed as a worst-case evaluation, assuming a fully active malicious server capable of modifying model architecture and exploiting gradient updates. While extreme, this assumption is appropriate for international or cross-consortium federated settings, where governance may be inconsistent and insider threats cannot be ruled out.</p><p>Several constraints limit the generalizability of our findings:</p><list list-type="order"><list-item><p>Dataset scope. We used publicly available radiology corpora (dischargesum and MIMIC-CXR), which lack multimodal components (eg, images), free-text notes, and operational metadata such as timestamps or identifiers. Future evaluations should test gradient inversion on richer, production-like datasets to better estimate clinical leakage risk.</p></list-item><list-item><p>Within-batch sequence correlation. Reports were split into nonoverlapping 32-token windows; multiple windows from the same report may co-occur within a single training batch. This violates the IID assumption implicit in our bin-collision analysis and may either inflate or deflate empirical reconstruction success relative to the theoretical per-token collision rate. A correlation-aware extension of the analytic inversion bound is left to future work.</p></list-item><list-item><p>Reference-vocabulary overlap scope. The MedGemma NER analysis was applied across all 3 datasets (discharge summaries, diagnostic reports, and MIMIC-CXR) over 5 random seeds (<xref ref-type="table" rid="table1">Table 1</xref>); overlap was high and statistically indistinguishable across tokenizers in every dataset. Because it is based on exact surface-form matching, it is a lower bound on semantic recovery, and because it is measured against a corpus-level reference vocabulary rather than the specific source report, it is an upper bound on source-aligned recovery.</p></list-item><list-item><p>Single reproducible experiment. All reported values&#x2014;including the paired <italic>t</italic> tests in <xref ref-type="supplementary-material" rid="app4">Multimedia Appendix 4</xref>&#x2014;derive from one reproducible execution of the experimental grid, run under a single fixed configuration with the same random seeds throughout; there is no separate primary or verification experiment. The complete code and per-seed outputs are released, and every reported value can be reproduced by reexecuting the released grid.</p></list-item><list-item><p>Retrospective attack focus. This study focused solely on post hoc attack analysis. No privacy-preserving defenses (eg, differential privacy, secure aggregation, and anomaly detection) were implemented or evaluated. While this allows for a clean assessment of leakage potential, it leaves open questions about mitigation feasibility and deployment cost.</p></list-item><list-item><p>Tokenizer isolation versus full model effects. We controlled for model architecture to isolate the effect of tokenization on inversion risk. While informative, real-world systems often use both tokenizer and model coadaptation. Thus, privacy risks may be higher or lower depending on the full pipeline design.</p></list-item><list-item><p>Data partitioning. Our experiments use an approximately IID partition across 6 simulated clients. Real-world cross-institutional federated deployments are typically non-IID&#x2014;site-specific protocols, demographic skew, and specialty-specific report distributions can introduce heterogeneity that affects gradient sparsity and consequently inversion vulnerability. The direction of this effect is not a priori clear (non-IID gradients carry stronger per-client signal but also higher variance), and a systematic non-IID evaluation is left for future work.</p></list-item><list-item><p>Sequence length. The 32-token regime adopted here is shorter than typical clinical narrative passages. Longer sequences increase the input dimensionality m&#x00D7;L of each per-sample reconstruction, which raises the number of bins k required for unambiguous disentanglement and thereby reduces single-step reconstruction success in practice. Our reported leakage rates are therefore an upper bound for the 32-token regime; longer-sequence regimes are expected to be harder to attack with a single fixed k, all else equal. Quantitatively, the expected per-bin occupancy scales as B&#x00D7;L/k, so an adversary can restore single-occupancy&#x2014;and hence attack efficacy&#x2014;by scaling k approximately linearly with L (eg, moving from L=32 to L=512 requires roughly a 16-fold increase in k). This is a linear resource cost rather than an exponential barrier, so longer contexts do not intrinsically prevent closed-form inversion; the practical brake is detectability, since k bins enlarge the injected linear probe by approximately k&#x00D7;(m+1) parameters, making the larger k required for long contexts correspondingly more conspicuous to the parameter-count attestation we recommend below. Extending the analysis to longer-sequence training is left for future work.</p></list-item><list-item><p>Client count. Our experiments fix the simulated client count at 6, reflecting a typical small-consortium clinical collaboration. The analytic gradient-inversion attack characterized in this study operates on a single client&#x2019;s gradient at a single round, and the per-client per-round recovery rate is independent of the number of other clients participating in the federation. Aggregation effects across clients (which arise specifically under cryptographic secure aggregation or weighted averaging schemes that obscure individual client updates) are absent from this evaluation by design and are an explicit target for future defense-oriented work.</p></list-item><list-item><p>Probe placement. The imprint module is inserted immediately before the positional embedding so that token-level representations are exposed prior to any token-mixing layer (positional addition and attention). Placement at later layers would conflate tokenizer-segmentation effects with the dynamics of attention and feed-forward mixing, and would not be informative for the central question of this study. A systematic placement ablation across preembedding, postembedding, and postattention positions&#x2014;holding the tokenizer fixed&#x2014;would isolate the contribution of placement itself and is identified here as a target for future work.</p></list-item></list><p>Taken together, these limitations do not diminish the central contribution highlighting a structural vulnerability in federated clinical NLP but emphasize the need for prospective, defense-integrated evaluations in more realistic medical AI pipelines.</p></sec><sec id="s4-9"><title>Future Work</title><p>Several extensions to this study are warranted. First, systematic ablation of the imprint module&#x2019;s placement (preembedding, postembedding, and postattention) would empirically validate the design rationale used here. Second, evaluation of differential privacy as a defense&#x2014;including utility-privacy trade-off curves at multiple &#x03B5; values, per-example clipping schedules, and interaction with secure aggregation&#x2014;is a substantial study in its own right and is the explicit subject of our planned follow-up work. Third, extension of the reference-vocabulary overlap analysis to source-report-aligned entity recovery and to non-IID federated partitions would strengthen the external validity of the leakage estimates reported here.</p></sec><sec id="s4-10"><title>Conclusions</title><p>FL holds great promise for enabling collaborative, multisite radiology AI without centralizing patient records. However, our gradient-inversion experiments demonstrate that current FL pipelines can leak substantial portions of patient text, raising significant concerns under HIPAA and GDPR standards. In worst-case scenarios, up to approximately 75% of sentences were exactly reconstructed from shared gradients, and approximately 75% of clinical entities, including diagnoses, procedures, and medications, were recovered from the reconstructed text.</p><p>Importantly, domain-adapted tokenizers such as RadBERT, while improving semantic fidelity, did not show significantly greater vulnerability to data reconstruction attacks than general-purpose alternatives such as GPT-2 and LLaMA-2. The privacy risk is substantial for all tokenizers and is governed by batch size rather than tokenizer choice.</p><p>To ensure safe deployment of federated clinical models, radiology departments and AI developers should rigorously evaluate privacy vulnerabilities in real-world settings, balancing utility with exposure risk. Regulatory bodies and standardization agencies may consider establishing clear audit pathways and certification frameworks tailored to the unique threats posed by clinical text.</p></sec></sec></body><back><ack><p>The authors thank Maximilian Zenk (Division of Medical Image Computing, DKFZ) for his valuable insights during manuscript preparation. We also appreciate the guidance and prior work of Jonas Geiping and Liam Fowl in developing federated transformer methods. We acknowledge the developers of PyTorch and Hugging Face for their machine learning frameworks, and the curators of the Dischargesum dataset for enabling access to radiology report data. Generative AI tools (Anthropic Claude) were used to assist with language editing, grammar checking, and formatting during manuscript revision. All scientific content, analyses, and conclusions are the authors' own, and the authors take full responsibility for the integrity of the manuscript.</p></ack><notes><sec><title>Funding</title><p>This work was partially supported by the PrivateAIM project, funded under the Medical Informatics Initiative by the German Federal Ministry of Education and Research (funding code 01ZZ2316A-O). The funder had no involvement in the study design, data collection, analysis, interpretation, or writing of the manuscript.</p></sec><sec><title>Data Availability</title><p>The radiology report corpora analyzed in this study are publicly available: the Dischargesum dataset and the MIMIC-CXR via PhysioNet. The code implementing the federated learning experiments and gradient-inversion attacks, including configuration files and per-seed output metrics, is publicly available at GitHub [<xref ref-type="bibr" rid="ref48">48</xref>]. All other data generated or analyzed during this study are included in the main manuscript.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: SP, RF</p><p>Data curation: SP</p><p>Formal analysis: SP</p><p>Investigation: SP</p><p>Methodology: SP</p><p>Project administration: SP</p><p>Resources: SS, RF</p><p>Software: SP</p><p>Supervision: SS, KM-H, RF</p><p>Validation: SP</p><p>Visualization: SP, AMM</p><p>Writing &#x2013; original draft: SP</p><p>Writing &#x2013; review and editing: SP, AMM, DB, SS, KM-H, RF</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">BLEU</term><def><p>bilingual evaluation understudy</p></def></def-item><def-item><term id="abb2">DKFZ</term><def><p>German Cancer Research Center</p></def></def-item><def-item><term id="abb3">DLG</term><def><p>deep leakage from gradients</p></def></def-item><def-item><term id="abb4">DP</term><def><p>differential privacy</p></def></def-item><def-item><term id="abb5">FAISS</term><def><p>Facebook AI Similarity Search</p></def></def-item><def-item><term id="abb6">FL</term><def><p>federated learning</p></def></def-item><def-item><term id="abb7">GDPR</term><def><p>General Data Protection Regulation</p></def></def-item><def-item><term id="abb8">HIPAA</term><def><p>Health Insurance Portability and Accountability Act</p></def></def-item><def-item><term id="abb9">iDLG</term><def><p>improved deep leakage from gradients</p></def></def-item><def-item><term id="abb10">IID</term><def><p>independent and identically distributed</p></def></def-item><def-item><term id="abb11">LCS</term><def><p>longest common subsequence</p></def></def-item><def-item><term id="abb12">LLM</term><def><p>large language model</p></def></def-item><def-item><term id="abb13">MIMIC-CXR</term><def><p>Medical Information Mart for Intensive Care Chest X-Ray</p></def></def-item><def-item><term id="abb14">NER</term><def><p>named-entity recognition</p></def></def-item><def-item><term id="abb15">NLP</term><def><p>natural language processing</p></def></def-item><def-item><term id="abb16">ROUGE-L</term><def><p>recall-oriented understudy for gisting evaluation</p></def></def-item><def-item><term id="abb17">S-BLEU</term><def><p>sentence-level bilingual evaluation understudy</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Castillo</surname><given-names>C</given-names> </name><name name-style="western"><surname>Steffens</surname><given-names>T</given-names> </name><name name-style="western"><surname>Sim</surname><given-names>L</given-names> </name><name name-style="western"><surname>Caffery</surname><given-names>L</given-names> </name></person-group><article-title>The effect of clinical information on radiology reporting: a systematic review</article-title><source>J Med Radiat Sci</source><year>2021</year><month>03</month><volume>68</volume><issue>1</issue><fpage>60</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1002/jmrs.424</pub-id><pub-id pub-id-type="medline">32870580</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Thirunavukarasu</surname><given-names>AJ</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSJ</given-names> </name><name name-style="western"><surname>Elangovan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gutierrez</surname><given-names>L</given-names> </name><name name-style="western"><surname>Tan</surname><given-names>TF</given-names> </name><name name-style="western"><surname>Ting</surname><given-names>DSW</given-names> </name></person-group><article-title>Large language models in medicine</article-title><source>Nat Med</source><year>2023</year><month>08</month><volume>29</volume><issue>8</issue><fpage>1930</fpage><lpage>1940</lpage><pub-id pub-id-type="doi">10.1038/s41591-023-02448-8</pub-id><pub-id pub-id-type="medline">37460753</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Radford</surname><given-names>A</given-names> </name><name name-style="western"><surname>Narasimhan</surname><given-names>K</given-names> </name></person-group><article-title>Improving language understanding by generative pre-training</article-title><source>Semantic Scholar</source><year>2018</year><access-date>2026-08-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://api.semanticscholar.org/CorpusID:49313245">https://api.semanticscholar.org/CorpusID:49313245</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Veen</surname><given-names>DV</given-names> </name><name name-style="western"><surname>Uden</surname><given-names>CV</given-names> </name><name name-style="western"><surname>Blankemeier</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Clinical text summarization: adapting large language models can outperform human experts</article-title><source>Res Sq</source><comment>Preprint posted online on  Oct 30, 2023</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-3483777/v1</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lecler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Duron</surname><given-names>L</given-names> </name><name name-style="western"><surname>Soyer</surname><given-names>P</given-names> </name></person-group><article-title>Revolutionizing radiology with GPT-based models: current applications, future possibilities and limitations of ChatGPT</article-title><source>Diagn Interv Imaging</source><year>2023</year><month>06</month><volume>104</volume><issue>6</issue><fpage>269</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1016/j.diii.2023.02.003</pub-id><pub-id pub-id-type="medline">36858933</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>D</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>B</given-names> </name></person-group><article-title>Large language models in summarizing radiology report impressions for lung cancer in Chinese: evaluation study</article-title><source>J Med Internet Res</source><year>2025</year><month>04</month><day>3</day><volume>27</volume><fpage>e65547</fpage><pub-id pub-id-type="doi">10.2196/65547</pub-id><pub-id pub-id-type="medline">40179389</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Evaluating large language models for automated reporting and data systems categorization: cross-sectional study</article-title><source>JMIR Med Inform</source><year>2024</year><month>07</month><day>17</day><volume>12</volume><fpage>e55799</fpage><pub-id pub-id-type="doi">10.2196/55799</pub-id><pub-id pub-id-type="medline">39018102</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shool</surname><given-names>S</given-names> </name><name name-style="western"><surname>Adimi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saboori Amleshi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bitaraf</surname><given-names>E</given-names> </name><name name-style="western"><surname>Golpira</surname><given-names>R</given-names> </name><name name-style="western"><surname>Tara</surname><given-names>M</given-names> </name></person-group><article-title>A systematic review of large language model (LLM) evaluations in clinical medicine</article-title><source>BMC Med Inform Decis Mak</source><year>2025</year><month>03</month><day>7</day><volume>25</volume><issue>1</issue><fpage>117</fpage><pub-id pub-id-type="doi">10.1186/s12911-025-02954-4</pub-id><pub-id pub-id-type="medline">40055694</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="web"><article-title>The HIPAA privacy rule</article-title><source>US Department of Health and Human Services</source><access-date>2027-08-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.hhs.gov/hipaa/for-professionals/privacy/index.html">https://www.hhs.gov/hipaa/for-professionals/privacy/index.html</ext-link></comment></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="web"><article-title>General Data Protection Regulation (GDPR)</article-title><source>GDPR.au</source><access-date>2026-08-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://gdpr.eu/tag/gdpr/">https://gdpr.eu/tag/gdpr/</ext-link></comment></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>McMahan</surname><given-names>B</given-names> </name><name name-style="western"><surname>Moore</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ramage</surname><given-names>D</given-names> </name><name name-style="western"><surname>Hampson</surname><given-names>S</given-names> </name><name name-style="western"><surname>Arcas</surname><given-names>BA</given-names> </name></person-group><article-title>Communication-efficient learning of deep networks from decentralized data</article-title><conf-name>20th International Conference on Artificial Intelligence and Statistics (AISTATS) 2017</conf-name><conf-date>Apr 20-22, 2017</conf-date><conf-loc>Fort Lauderdale, FL, USA</conf-loc><fpage>1273</fpage><lpage>1282</lpage><pub-id pub-id-type="doi">10.48550/arXiv.1602.05629</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Khor</surname><given-names>HG</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Continually tuning a large language model for multi-domain radiology report generation</article-title><source>Int Conf Med Image Comput Comput-Assist Interv</source><year>2024</year><fpage>177</fpage><lpage>187</lpage><pub-id pub-id-type="doi">10.1007/978-3-031-72086-4_17</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bhayana</surname><given-names>R</given-names> </name></person-group><article-title>Chatbots and large language models in radiology: a practical primer for clinical and research applications</article-title><source>Radiology</source><year>2024</year><month>01</month><volume>310</volume><issue>1</issue><fpage>e232756</fpage><pub-id pub-id-type="doi">10.1148/radiol.232756</pub-id><pub-id pub-id-type="medline">38226883</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akinci D&#x2019;Antonoli</surname><given-names>T</given-names> </name><name name-style="western"><surname>Stanzione</surname><given-names>A</given-names> </name><name name-style="western"><surname>Bluethgen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Large language models in radiology: fundamentals, applications, ethical considerations, risks, and future directions</article-title><source>Diagn Interv Radiol</source><year>2024</year><month>03</month><day>1</day><volume>30</volume><issue>2</issue><fpage>80</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.4274/dir.2023.232417</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wong</surname><given-names>IN</given-names> </name><name name-style="western"><surname>Monteiro</surname><given-names>O</given-names> </name><name name-style="western"><surname>Baptista-Hon</surname><given-names>DT</given-names> </name><etal/></person-group><article-title>Leveraging foundation and large language models in medical artificial intelligence</article-title><source>Chin Med J (Engl)</source><year>2024</year><month>11</month><day>5</day><volume>137</volume><issue>21</issue><fpage>2529</fpage><lpage>2539</lpage><pub-id pub-id-type="doi">10.1097/CM9.0000000000003302</pub-id><pub-id pub-id-type="medline">39497256</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Parampottupadam</surname><given-names>S</given-names> </name><name name-style="western"><surname>Floca</surname><given-names>R</given-names> </name><name name-style="western"><surname>Bounias</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Client security alone fails in federated learning: 2D and 3D attack insights</article-title><source>Medical Information Computing MImA EMERGE 2024 Communications in Computer and Information Science</source><year>2024</year><access-date>2026-01-12</access-date><publisher-name>Springer</publisher-name><fpage>235</fpage><lpage>244</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/978-3-031-79103-1_24">https://doi.org/10.1007/978-3-031-79103-1_24</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Brauneck</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schmalhorst</surname><given-names>L</given-names> </name><name name-style="western"><surname>Kazemi Majdabadi</surname><given-names>MM</given-names> </name><etal/></person-group><article-title>Federated machine learning, privacy-enhancing technologies, and data protection laws in medical research: scoping review</article-title><source>J Med Internet Res</source><year>2023</year><month>03</month><day>30</day><volume>25</volume><fpage>e41588</fpage><pub-id pub-id-type="doi">10.2196/41588</pub-id><pub-id pub-id-type="medline">36995759</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Personalized and privacy-preserving federated heterogeneous medical image analysis with PPPML-HMI</article-title><source>Comput Biol Med</source><year>2024</year><month>02</month><volume>169</volume><fpage>107861</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2023.107861</pub-id><pub-id pub-id-type="medline">38141449</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaissis</surname><given-names>G</given-names> </name><name name-style="western"><surname>Ziller</surname><given-names>A</given-names> </name><name name-style="western"><surname>Passerat-Palmbach</surname><given-names>J</given-names> </name><etal/></person-group><article-title>End-to-end privacy preserving deep learning on multi-institutional medical imaging</article-title><source>Nat Mach Intell</source><year>2021</year><volume>3</volume><issue>6</issue><fpage>473</fpage><lpage>484</lpage><pub-id pub-id-type="doi">10.1038/s42256-021-00337-8</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hatamizadeh</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>H</given-names> </name><name name-style="western"><surname>Molchanov</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Do gradient inversion attacks make federated learning unsafe?</article-title><source>IEEE Trans Med Imaging</source><year>2023</year><month>07</month><volume>42</volume><issue>7</issue><fpage>2044</fpage><lpage>2056</lpage><pub-id pub-id-type="doi">10.1109/TMI.2023.3239391</pub-id><pub-id pub-id-type="medline">37021996</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Akinci D&#x2019;Antonoli</surname><given-names>T</given-names> </name><name name-style="western"><surname>Bluethgen</surname><given-names>C</given-names> </name></person-group><article-title>A new era of text mining in radiology with privacy-preserving LLMs</article-title><source>Radiol Artif Intell</source><year>2024</year><month>07</month><volume>6</volume><issue>4</issue><fpage>e240261</fpage><pub-id pub-id-type="doi">10.1148/ryai.240261</pub-id><pub-id pub-id-type="medline">38900034</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tinn</surname><given-names>R</given-names> </name><name name-style="western"><surname>Cheng</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Domain-specific language model pretraining for biomedical natural language processing</article-title><source>ACM Trans Comput Healthcare</source><year>2022</year><month>01</month><day>31</day><volume>3</volume><issue>1</issue><fpage>1</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.1145/3458754</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Petzold</surname><given-names>LR</given-names> </name></person-group><article-title>AlpaCare: instruction-tuned large language models for medical application</article-title><source>arXiv</source><access-date>2026-08-24</access-date><comment>Preprint posted online on  Oct 23, 2023</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2310.14558">https://arxiv.org/abs/2310.14558</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yan</surname><given-names>A</given-names> </name><name name-style="western"><surname>McAuley</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>X</given-names> </name><etal/></person-group><article-title>RadBERT: adapting transformer-based language models to radiology</article-title><source>Radiol Artif Intell</source><year>2022</year><month>07</month><volume>4</volume><issue>4</issue><fpage>e210258</fpage><pub-id pub-id-type="doi">10.1148/ryai.210258</pub-id><pub-id pub-id-type="medline">35923376</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Solaiman</surname><given-names>I</given-names> </name><name name-style="western"><surname>Brundage</surname><given-names>M</given-names> </name><name name-style="western"><surname>Clark</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Release strategies and the social impacts of language models</article-title><source>arXiv</source><access-date>2026-08-24</access-date><comment>Preprint posted online on  Aug 24, 2019</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1908.09203">https://arxiv.org/abs/1908.09203</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Touvron</surname><given-names>H</given-names> </name><name name-style="western"><surname>Martin</surname><given-names>L</given-names> </name><name name-style="western"><surname>Stone</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Llama 2: open foundation and fine-tuned chat models</article-title><source>arXiv</source><access-date>2026-08-24</access-date><comment>Preprint posted online on  Jul 18, 2023</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2307.09288">https://arxiv.org/abs/2307.09288</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><article-title>Dischargesum dataset</article-title><source>Hugging Face</source><year>2024</year><access-date>2026-08-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://huggingface.co/dischargesum">https://huggingface.co/dischargesum</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>AEW</given-names> </name><name name-style="western"><surname>Pollard</surname><given-names>TJ</given-names> </name><name name-style="western"><surname>Greenbaum</surname><given-names>NR</given-names> </name><etal/></person-group><article-title>MIMIC-CXR-JPG, a large publicly available database of labeled chest radiographs</article-title><access-date>2026-08-24</access-date><comment>Preprint posted online on  Nov 14, 2019</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1901.07042">https://arxiv.org/abs/1901.07042</ext-link></comment></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kairouz</surname><given-names>P</given-names> </name><name name-style="western"><surname>McMahan</surname><given-names>HB</given-names> </name><name name-style="western"><surname>Avent</surname><given-names>B</given-names> </name></person-group><article-title>Advances and open problems in federated learning</article-title><source>Found Trends Mach Learn</source><year>2021</year><month>06</month><day>23</day><volume>14</volume><issue>1-2</issue><fpage>1</fpage><lpage>210</lpage><pub-id pub-id-type="doi">10.1561/2200000083</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Fowl</surname><given-names>L</given-names> </name><name name-style="western"><surname>Geiping</surname><given-names>J</given-names> </name><name name-style="western"><surname>Czaja</surname><given-names>W</given-names> </name><name name-style="western"><surname>Goldblum</surname><given-names>M</given-names> </name><name name-style="western"><surname>Goldstein</surname><given-names>T</given-names> </name></person-group><article-title>Robbing the fed: directly obtaining private data in federated learning with modified models</article-title><source>arXiv</source><access-date>2026-08-30</access-date><comment>Preprint posted online on  Oct 25, 2022</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2110.13057">https://arxiv.org/abs/2110.13057</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Geiping</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bauermeister</surname><given-names>H</given-names> </name><name name-style="western"><surname>Dr&#x00F6;ge</surname><given-names>H</given-names> </name><name name-style="western"><surname>Moeller</surname><given-names>M</given-names> </name></person-group><article-title>Inverting gradients &#x2014; how easy is it to break privacy in federated learning?</article-title><access-date>2026-08-30</access-date><conf-name>34th International Conference on Neural Information Processing Systems (NeurIPS 2020)</conf-name><conf-date>Dec 6-12, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2003.14053">https://arxiv.org/abs/2003.14053</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><etal/></person-group><article-title>TAG: gradient attack on transformer-based language models</article-title><source>arXiv</source><comment>Preprint posted online on  Mar 11, 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2103.06819</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Sellergren</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kazemzadeh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jaroensri</surname><given-names>T</given-names> </name><etal/></person-group><article-title>MedGemma technical report</article-title><source>arXiv</source><comment>Preprint posted online on  Apr 6, 2026</comment><pub-id pub-id-type="doi">10.48550/arXiv.2507.05201</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Papineni</surname><given-names>K</given-names> </name><name name-style="western"><surname>Roukos</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ward</surname><given-names>T</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>WJ</given-names> </name></person-group><article-title>BLEU: a method for automatic evaluation of machine translation</article-title><source>Proceedings of the 40th Annual Meeting of the Association for Computational Linguistics</source><year>2002</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>311</fpage><lpage>318</lpage><pub-id pub-id-type="doi">10.3115/1073083.1073135</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Delbrouck</surname><given-names>JB</given-names> </name><name name-style="western"><surname>Varma</surname><given-names>M</given-names> </name><name name-style="western"><surname>Chambon</surname><given-names>P</given-names> </name><name name-style="western"><surname>Langlotz</surname><given-names>C</given-names> </name></person-group><article-title>Overview of the RadsSum23 shared task on multi-modal and multi-anatomical radiology report summarization 22nd workshop biomed nat lang process bionlp shar tasks toronto</article-title><source>Proceedings of the 22nd Workshop on Biomedical Natural Language Processing and BioNLP Shared Tasks</source><year>2023</year><publisher-name>Association for Computational Linguistics</publisher-name><fpage>478</fpage><lpage>482</lpage><pub-id pub-id-type="doi">10.18653/v1/2023.bionlp-1.45</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>CY</given-names> </name></person-group><article-title>ROUGE: a package for automatic evaluation of summaries</article-title><access-date>2026-02-20</access-date><conf-name>Text Summarization Branches Out</conf-name><conf-loc>Barcelona, Spain</conf-loc><fpage>74</fpage><lpage>81</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/W04-1013/">https://aclanthology.org/W04-1013/</ext-link></comment></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Boenisch</surname><given-names>F</given-names> </name><name name-style="western"><surname>Dziedzic</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schuster</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shamsabadi</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Shumailov</surname><given-names>I</given-names> </name><name name-style="western"><surname>Papernot</surname><given-names>N</given-names> </name></person-group><article-title>When the curious abandon honesty: federated learning is not private</article-title><conf-name>2023 IEEE 8th European Symposium on Security and Privacy (EuroS&#x0026;P)</conf-name><conf-date>Jul 3-7, 2023</conf-date><pub-id pub-id-type="doi">10.1109/EuroSP57164.2023.00020</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Han</surname><given-names>S</given-names> </name></person-group><article-title>Deep leakage from gradients</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 21, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1906.08935</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>B</given-names> </name><name name-style="western"><surname>Mopuri</surname><given-names>KR</given-names> </name><name name-style="western"><surname>Bilen</surname><given-names>H</given-names> </name></person-group><article-title>IDLG: improved deep leakage from gradients</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 8, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2001.02610</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Dwork</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bugliesi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Preneel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sassone</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wegener</surname><given-names>I</given-names> </name></person-group><article-title>Differential privacy</article-title><source>Autom Lang Program</source><year>2006</year><access-date>2025-11-12</access-date><publisher-name>Springer</publisher-name><fpage>1</fpage><lpage>12</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://link.springer.com/chapter/10.1007/11787006_1">https://link.springer.com/chapter/10.1007/11787006_1</ext-link></comment></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Bonawitz</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ivanov</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kreuter</surname><given-names>B</given-names> </name><etal/></person-group><article-title>Practical secure aggregation for privacy-preserving machine learning</article-title><source>CCS &#x2019;17: Proceedings of the 2017 ACM SIGSAC Conference on Computer and Communications Security</source><year>2017</year><publisher-name>Association for Computing Machinery</publisher-name><fpage>1175</fpage><lpage>1191</lpage><pub-id pub-id-type="doi">10.1145/3133956.3133982</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaabachi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Despraz</surname><given-names>J</given-names> </name><name name-style="western"><surname>Meurers</surname><given-names>T</given-names> </name><etal/></person-group><article-title>A scoping review of privacy and utility metrics in medical synthetic data</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>27</day><volume>8</volume><issue>1</issue><fpage>60</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01359-3</pub-id><pub-id pub-id-type="medline">39870798</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Kang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Q</given-names> </name></person-group><article-title>Trading off privacy, utility, and efficiency in federated learning</article-title><source>ACM Trans Intell Syst Technol</source><year>2023</year><month>12</month><day>31</day><volume>14</volume><issue>6</issue><fpage>1</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1145/3595185</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Akkus</surname><given-names>A</given-names> </name><name name-style="western"><surname>Aghdam</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Generated data with fake privacy: hidden dangers of fine-tuning large language models on generated data</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 29, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2409.11423</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Stadler</surname><given-names>T</given-names> </name><name name-style="western"><surname>Oprisanu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Troncoso</surname><given-names>C</given-names> </name></person-group><article-title>Synthetic data &#x2014; anonymisation groundhog day</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 24, 2022</comment><pub-id pub-id-type="doi">10.48550/arXiv.2011.07018</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jayaraman</surname><given-names>B</given-names> </name><name name-style="western"><surname>Evans</surname><given-names>D</given-names> </name></person-group><article-title>Evaluating differentially private machine learning in practice</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 12, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1902.08874</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hosseini</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Sikaroudi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Babaei</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tizhoosh</surname><given-names>HR</given-names> </name></person-group><article-title>Cluster based secure multi-party computation in federated learning for histopathology images</article-title><source>arXiv</source><comment>Preprint posted online on  Aug 21, 2022</comment><pub-id pub-id-type="doi">10.1007/978-3-031-18523-6_11</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="web"><article-title>Privacy leakage in federated learning in radiology reports</article-title><source>GitHub</source><access-date>2026-08-24</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/santhoshcameo/Privacy-Leakage-in-Federated-Learning-in-Radiology-Reports">https://github.com/santhoshcameo/Privacy-Leakage-in-Federated-Learning-in-Radiology-Reports</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Distribution of exact sentence reconstruction accuracy across datasets, tokenizers, and batch sizes (box plots with the 5 individual runs overlaid).</p><media xlink:href="medinform_v14i1e88390_app1.docx" xlink:title="DOCX File, 94 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Distribution of recall-oriented understudy for gisting evaluation scores across datasets, tokenizers, and batch sizes (box plots with the 5 individual runs overlaid).</p><media xlink:href="medinform_v14i1e88390_app2.docx" xlink:title="DOCX File, 86 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>95% CIs for exact sentence reconstruction accuracy, computed from the per-cell SDs (Student <italic>t</italic> test, <italic>t</italic><sub>4</sub>=2.776). Means are reproduced exactly from <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><media xlink:href="medinform_v14i1e88390_app3.docx" xlink:title="DOCX File, 12 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>Pairwise tokenizer paired <italic>t</italic> tests on exact sentence accuracy across the 5 seeds (<italic>df</italic>=4). &#x0394; is the mean per-seed difference in percentage points.</p><media xlink:href="medinform_v14i1e88390_app4.docx" xlink:title="DOCX File, 26 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Training details and reproducibility configuration. Hyperparameters and infrastructure settings used for all experiments.</p><media xlink:href="medinform_v14i1e88390_app5.docx" xlink:title="DOCX File, 11 KB"/></supplementary-material><supplementary-material id="app6"><label>Multimedia Appendix 6</label><p>Exact sentence reconstruction accuracy (%) by tokenizer, dataset, and batch size (mean, SD over 5 seeds).</p><media xlink:href="medinform_v14i1e88390_app6.docx" xlink:title="DOCX File, 11 KB"/></supplementary-material><supplementary-material id="app7"><label>Multimedia Appendix 7</label><p>Mean reconstruction scores for sentence-level bilingual evaluation understudy (S) and recall-oriented understudy for gisting evaluation (R) across batch sizes for 3 datasets and tokenizers (mean of 5 seeds).</p><media xlink:href="medinform_v14i1e88390_app7.docx" xlink:title="DOCX File, 11 KB"/></supplementary-material></app-group></back></article>