<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e97661</article-id><article-id pub-id-type="doi">10.2196/97661</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Quantifying the Impact of Anonymization-Induced Clinical Data Quality Loss: Methodological Quantitative Case Study Using Primary Diagnosis Codes and Hospital Length of Stay</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Kamdje Wabo</surname><given-names>Gaetan</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sokolowski</surname><given-names>Piotr Pawel</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Jannesari Ladani</surname><given-names>Mahboubeh</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hagmann</surname><given-names>Michael</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ganslandt</surname><given-names>Thomas</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Siegel</surname><given-names>Fabian</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Biomedical Informatics, Mannheim Institute for Intelligent Systems in Medicine (MIISM), Medical Faculty of Mannheim, University of Heidelberg</institution><addr-line>Theodor-Kutzer-Ufer 1&#x2013;3, House 3, Floor 4</addr-line><addr-line>Mannheim</addr-line><country>Germany</country></aff><aff id="aff2"><institution>Institute of Medical Informatics, Biometry and Epidemiology, Friedrich-Alexander-Universit&#x00E4;t Erlangen-N&#x00FC;rnberg</institution><addr-line>Erlangen</addr-line><addr-line>Bavaria</addr-line><country>Germany</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Cohen</surname><given-names>Ariel</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Gaetan Kamdje Wabo, MSc, Department of Biomedical Informatics, Mannheim Institute for Intelligent Systems in Medicine (MIISM), Medical Faculty of Mannheim, University of Heidelberg, Theodor-Kutzer-Ufer 1&#x2013;3, House 3, Floor 4, Mannheim, 68167, Germany, 49 621 383 8088; <email>gaetankamdje.wabo@medma.uni-heidelberg.de</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>11</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e97661</elocation-id><history><date date-type="received"><day>08</day><month>04</month><year>2026</year></date><date date-type="rev-recd"><day>06</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>08</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Gaetan Kamdje Wabo, Piotr Pawel Sokolowski, Mahboubeh Jannesari Ladani, Michael Hagmann, Thomas Ganslandt, Fabian Siegel. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 11.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e97661"/><abstract><sec><title>Background</title><p>The secondary use of electronic health record data requires robust privacy protection. k-Anonymity is widely used to enable data sharing by ensuring that each quasi-identifier combination occurs in at least k records; yet, its analytical impact on clinically meaningful structures remains insufficiently characterized, particularly for the combination of record suppression and microaggregation that arises when a numeric attribute lacks a natural generalization hierarchy. A further gap is that anonymization tools report internal information-loss values but do not signal the downstream distributional and inferential distortions these transformations introduce.</p></sec><sec><title>Objective</title><p>This study evaluated the analytical footprint of k-anonymity at k=5, 10, and 15 on 2 core data elements in retrospective hospital research: primary <italic>International Classification of Diseases, 10th Revision, German Modification</italic> (ICD-10-GM) diagnosis codes, and hospital length of stay (LOS). It aimed to determine and quantify whether anonymization introduces meaningful distortions not captured by the anonymization tool itself, and whether diagnosis-specific LOS patterns remain reproducible after anonymization.</p></sec><sec sec-type="methods"><title>Methods</title><p>We analyzed 719,387 inpatient encounters from University Hospital Mannheim from 2010 to 2024. Anonymization was performed with the ARX tool. It used record suppression and microaggregation. Distributional distortion was assessed with the Kolmogorov-Smirnov <italic>D</italic> statistic, quantile shifts, IQR changes, and tail changes. Categorical fidelity was assessed with the Jaccard coefficient and Cramer <italic>V</italic>. Inferential reproducibility was assessed with a 3-level linear mixed model. The model included random intercepts for the three-character diagnosis codes from the <italic>International Classification of Diseases</italic> (<italic>ICD-3</italic>) and patients. We compared the intraclass correlation coefficient and diagnosis-level effect concordance. Concordance was quantified using Spearman &#x03C1; and Lin concordance correlation coefficient, both with 95% CIs. A composite traffic-light verdict summarized the results.</p></sec><sec sec-type="results"><title>Results</title><p>ARX masked quasi-identifier cells rather than deleting rows; the proportion of encounters with a masked cell rose from 0.77% (k=5) to 2.62% (k=15), distributed almost uniformly across admission years. Kolmogorov-Smirnov <italic>D</italic> was stable at 0.147. Median LOS shifted by 1 day, and the SD declined by about 6.5 days, while the diagnosis-level mean changed considerably (absolute mean shift of &#x2212;1.81 days). Jaccard overlap fell from 0.624 to 0.421. The diagnosis intraclass correlation coefficient rose from 0.294 to 0.837, reflecting variance compression rather than improved signal. Best linear unbiased prediction rank concordance (Spearman &#x03C1; 0.964&#x2010;0.970) and aggregate magnitude agreement (Lin concordance correlation coefficient 0.959&#x2010;0.966) were high; yet, about 7% of low-signal diagnoses showed sign reversals. Most distributional change occurred at k=5.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>k-Anonymity preserved the ranking of diagnosis-specific LOS effects but altered distributional shape, individual effect magnitudes, and diagnostic vocabulary, none of which was flagged by the internal loss metric. Anonymized data of this type may support ordinal analyses but can mislead analyses requiring faithful variance structure, accurate absolute effects, or complete rare-diagnosis representation. A reporting checklist is provided to document these effects.</p></sec></abstract><kwd-group><kwd>ADAQI reporting checklist</kwd><kwd>Anonymization and Data Analysis Quality and Impact Reporting</kwd><kwd>data anonymization</kwd><kwd>k-anonymity</kwd><kwd>clinical data quality</kwd><kwd>length of stay</kwd><kwd>International Classification of Diseases</kwd><kwd>ICD codes</kwd><kwd>distributional fidelity</kwd><kwd>inferential reproducibility</kwd><kwd>linear mixed model</kwd><kwd>ARX Data Anonymization Tool</kwd><kwd>medical informatics</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>The increasing use of electronic health record data for secondary purposes has become a cornerstone of clinical research, health services planning, and AI development in medicine [<xref ref-type="bibr" rid="ref1">1</xref>]. This progress creates an inherent tension. Patient privacy must be safeguarded, while data quality must remain sufficient for valid statistical inference. The European General Data Protection Regulation [<xref ref-type="bibr" rid="ref2">2</xref>] codifies strict obligations for processing personal health information. Under the General Data Protection Regulation, anonymization may allow broader data sharing where the applied transformations make reidentification sufficiently unlikely in the relevant context.</p><p>Among the most widely adopted paradigms, k-anonymity, proposed by Sweeney [<xref ref-type="bibr" rid="ref3">3</xref>], guarantees that each combination of quasi-identifying attributes appears in at least k records. The ARX Data Anonymization Tool, developed by Prasser et al [<xref ref-type="bibr" rid="ref4">4</xref>], implements k-anonymity alongside l-diversity [<xref ref-type="bibr" rid="ref5">5</xref>], t-closeness [<xref ref-type="bibr" rid="ref6">6</xref>], and differential privacy [<xref ref-type="bibr" rid="ref7">7</xref>] within a unified open-source framework. ARX applies generalization hierarchies, record suppression, and microaggregation to achieve privacy guarantees while optimizing utility through configurable loss functions [<xref ref-type="bibr" rid="ref8">8</xref>]. Its application spans clinical trial data sharing [<xref ref-type="bibr" rid="ref9">9</xref>], pandemic cohort publication [<xref ref-type="bibr" rid="ref10">10</xref>], and multisite registry studies [<xref ref-type="bibr" rid="ref11">11</xref>]. Its broad acceptance in the research community is precisely what makes it a suitable target tool for a methodological case study, because findings obtained with ARX are relevant to a large body of existing and future work.</p><p>Despite the maturity of anonymization algorithms, systematic evaluation of downstream statistical consequences remains underdeveloped. Five principal approaches currently inform quality assessment. Information loss metrics embedded within tools capture transformation cost but not external analytical validity [<xref ref-type="bibr" rid="ref4">4</xref>]. Descriptive concordance methods compare summary statistics, with Ohno-Machado et al [<xref ref-type="bibr" rid="ref12">12</xref>] showing that suppression preserves means while degrading prediction. CI overlap approaches, as demonstrated by Pilgram et al [<xref ref-type="bibr" rid="ref13">13</xref>], evaluate analytical agreement under use-case&#x2013;specific configurations. Distributional divergence methods use statistical tests, applied for instance by Jakob et al [<xref ref-type="bibr" rid="ref10">10</xref>] to COVID-19 anonymization pipelines. Regression reproducibility frameworks, as used by Johann et al [<xref ref-type="bibr" rid="ref14">14</xref>], test whether clinical associations survive transformation. Recent work continues to confirm that privacy-enhancing transformations degrade analytical utility in ways that depend on the downstream task. Cohen et al [<xref ref-type="bibr" rid="ref15">15</xref>] quantified the effects of pseudonymization on 6 archetypal epidemiological analyses in a clinical data warehouse and showed that achieving low reidentification risk required modifications that compromised the reliability of specific statistics, and Kaabachi et al [<xref ref-type="bibr" rid="ref16">16</xref>] cataloged in a scoping review the heterogeneous privacy and utility metrics used across medical studies and the difficulty of identifying an optimal balance.</p><p>Each of these approaches illuminates one facet of the privacy-utility trade-off. However, their isolation limits explanatory power. Distributional fidelity metrics quantify what changed in the data structure but cannot determine whether those changes compromise downstream conclusions. Inferential reproducibility tests reveal whether conclusions changed but, without distributional context, cannot explain why or through which transformation mechanism. Combining both dimensions within a single evaluation approach can address this interpretive gap. Distortion patterns such as variance compression or vocabulary erosion, once identified through distributional analysis, can be directly linked to concordance or divergence in the statistical estimates derived from the transformed data. This gap is particularly relevant when anonymization relies on record suppression combined with microaggregation, a configuration that arises whenever generalization hierarchies cannot be specified for all quasi-identifiers. Numeric attributes such as hospital length of stay (LOS) lack a natural semantic hierarchy [<xref ref-type="bibr" rid="ref17">17</xref>], compelling tools such as ARX to apply microaggregation instead [<xref ref-type="bibr" rid="ref8">8</xref>]. The analytical consequences of this specific combination remain largely unquantified, as existing evaluations have focused predominantly on generalization-based pipelines [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>Against this methodological backdrop, 2 clinical data elements occupy a uniquely central position in hospital research. These are the primary diagnosis coded in the <italic>International Classification of Diseases</italic> (ICD) system and the hospital LOS. Primary ICD codes define case classification for diagnosis-related group reimbursement [<xref ref-type="bibr" rid="ref18">18</xref>], serve as epidemiological cohort criteria, and anchor outcome research. The LOS functions as a proxy for disease severity, resource consumption, and care efficiency [<xref ref-type="bibr" rid="ref19">19</xref>]. The association between diagnosis and LOS has been extensively investigated [<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref29">29</xref>]. Any anonymization-induced distortion of these attributes may directly threaten the validity of downstream analyses. Given this context, 3 research questions (RQs) guided this investigation.</p></sec><sec id="s1-2"><title>RQs</title><p>The 3 RQs are as follows:</p><list list-type="bullet"><list-item><p>RQ1 (distributional fidelity): To what extent does k-anonymity at k=5, 10, and 15 alter the distributional properties of primary <italic>International Classification of Diseases, 10th Revision, German Modification</italic> (ICD-10-GM) diagnosis codes and hospital LOS, as measured by descriptive divergence statistics (Kolmogorov-Smirnov [KS] <italic>D</italic>), complementary distributional descriptors (quantile shifts and IQR compression), Cramer <italic>V</italic>, and the Jaccard similarity coefficient?</p></list-item><list-item><p>RQ2 (inferential reproducibility): Does the diagnosis-associated variation in hospital LOS remain reproducible after anonymization? In clinical terms, if a researcher uses anonymized data to identify which diagnoses are associated with the longest hospital stays, will the resulting ranking agree with the ranking obtained from the original data, and will the estimated magnitude of these differences remain sufficiently stable for interpretation?</p></list-item><list-item><p>RQ3 (dose-response shape): Does data quality degrade gradually with increasing k, or does most of the change occur at the first anonymization step? The practical implication is whether choosing a higher k level (eg, k=15 instead of k=5) imposes a substantially greater analytical cost, or whether the initial transformation already accounts for the dominant share of the change.</p></list-item></list></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This methodological case study combines distributional impact assessment with inferential reproducibility testing on a large-scale hospital encounter dataset. The intent is not to model a comprehensive real-world data release, but to demonstrate, using a widely adopted anonymization tool under self-configured anonymity, transformation, and utility parameters, that meaningful distributional and inferential distortions can occur which the tool itself does not report. The analysis proceeds in 4 phases: cohort extraction and characterization, anonymization at 3 k-anonymity thresholds using ARX, distributional distortion quantification through a purpose-built comparison framework, and inferential reproducibility evaluation by fitting identical linear mixed model (LMM) specifications to original and anonymized data.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study received approval from Ethics Committee II, Medical Faculty Mannheim, Heidelberg University (ID: 2024&#x2010;885). A data protection impact assessment was positively evaluated. The local Use and Access Committee granted formal approval. All analytical datasets were processed and handled in deidentified form within the approved institutional governance framework.</p></sec><sec id="s2-3"><title>Data Collection and Cohort Description</title><p>For this evaluation, hospital encounter data were extracted from the clinical data warehouse of the Medical Data Integration Center at University Hospital Mannheim (Universit&#x00E4;tsmedizin Mannheim). The query targeted inpatient encounters of the study period. For the present analysis, the cohort was rederived directly from recorded admission and discharge dates, retaining only encounters whose admission date and discharge date both fell within January 1, 2010, and September 30, 2024 (revised data preparation in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). The resulting dataset comprised 5 data elements per encounter: a pseudonymous patient identifier, a unique encounter identifier, the admission year, the primary ICD-10-GM diagnosis code, and the LOS. LOS was computed as the difference in days between discharge and admission, with same-day encounters assigned a value of 1. Because the cohort was defined on the basis of present and valid admission and discharge dates, no encounter had a missing LOS value, and no encounter was excluded for that reason.</p></sec><sec id="s2-4"><title>Data Anonymization</title><p>Anonymization was performed using the ARX Data Anonymization Tool (version 3.9.2) [<xref ref-type="bibr" rid="ref4">4</xref>]. Three separate configurations were applied to the same source dataset (N=719,387 encounters), producing anonymized variants at k=5, k=10, and k=15. Secure Hash Algorithm 256-bit checksums verified each output. <xref ref-type="table" rid="table1">Table 1</xref> details the complete ARX configuration. Two quasi-identifiers were designated: the primary ICD diagnosis code (categorical) and LOS in days (integer). The patient identifier and admission year were retained outside the quasi-identifier set, the former to enable patient-level random effects and the latter to enable a temporal sensitivity analysis, and neither was anonymized. These 2 variables were not intended for external release.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Configuration parameters of the ARX Data Anonymization Tool (version 3.9.2) applied to 719,387 inpatient encounters.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Parameter</td><td align="left" valign="bottom">Setting</td></tr></thead><tbody><tr><td align="left" valign="top">Anonymization model</td><td align="left" valign="top">k-Anonymity</td></tr><tr><td align="left" valign="top">k values tested</td><td align="left" valign="top">5, 10, 15</td></tr><tr><td align="left" valign="top">Quasi-identifiers</td><td align="left" valign="top">Primary ICD-10-GM<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup> code, LOS<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup> (days)</td></tr><tr><td align="left" valign="top">Retained non&#x2013;quasi-identifiers</td><td align="left" valign="top">Patient ID, encounter ID, admission year (not anonymized and not intended for release)</td></tr><tr><td align="left" valign="top">ICD<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup> transformation</td><td align="left" valign="top">No generalization hierarchy (height 1, min=max = 0)</td></tr><tr><td align="left" valign="top">LOS transformation</td><td align="left" valign="top">Microaggregation (geometric mean, ordered)</td></tr><tr><td align="left" valign="top">Microaggregation cluster size</td><td align="left" valign="top">Determined by ARX optimizer per k level</td></tr><tr><td align="left" valign="top">Suppression limit</td><td align="left" valign="top">0.50 (maximum fraction of records suppressible)</td></tr><tr><td align="left" valign="top">Suppression mechanism</td><td align="left" valign="top">Cell-level masking of ICD and LOS jointly (token *)</td></tr><tr><td align="left" valign="top">Optimization weights</td><td align="left" valign="top">ICD=0.50, LOS=0.50</td></tr><tr><td align="left" valign="top">Aggregate loss function</td><td align="left" valign="top">Geometric mean</td></tr><tr><td align="left" valign="top">Attacker models</td><td align="left" valign="top">Prosecutor, journalist, marketer (simultaneous)</td></tr><tr><td align="left" valign="top">Risk thresholds</td><td align="left" valign="top">Calibrated inversely to k</td></tr><tr><td align="left" valign="top">Search strategy</td><td align="left" valign="top">Optimal (deterministic, search space size=1)</td></tr><tr><td align="left" valign="top">Software version</td><td align="left" valign="top">ARX 3.9.2</td></tr><tr><td align="left" valign="top">Verification</td><td align="left" valign="top">SHA-256<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup> checksums per output</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>ICD-10-GM: International Classification of Diseases, 10th Revision, German Modification.</p></fn><fn id="table1fn2"><p><sup>b</sup>LOS: length of stay (days).</p></fn><fn id="table1fn3"><p><sup>c</sup>ICD: International Classification of Diseases.</p></fn><fn id="table1fn4"><p><sup>d</sup>SHA-256: Secure Hash Algorithm 256-bit.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-5"><title>Rationale for the ARX Configuration</title><p>Two configuration choices warrant explicit justification because they directly governed the suppression behavior and the resulting distortion. First, no generalization hierarchy was applied to ICD codes, which were treated as a flat categorical attribute by setting the hierarchy height to 1 with minimum equal to maximum equal to 0 [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. This was deliberate. RQ1 measures how anonymization alters the full ICD code distribution at the level at which codes are originally recorded, and RQ2 evaluates whether diagnosis-associated LOS differences remain reproducible at the 3-character level after the transformation. Pregeneralization would have collapsed full codes (eg, E11.40, E11.41, and E11.49) into their parent category (E11) before anonymization, eliminating the variability that the suppression mechanism is intended to act upon. Suppression would then operate on an already-coarsened code set, and the comparison between original and anonymized data would reflect the combined impact of precoarsening and anonymization rather than the impact of the anonymization step itself. Treating each full code as an indivisible unit ensured that suppression remained the only mechanism acting on the categorical structure.</p><p>Second, the suppression limit was set to its maximum admissible value of 0.50, equal optimization weights of 0.50 were assigned to both quasi-identifiers, and information loss was aggregated using the geometric mean. The equal weighting reflects the absence of an a priori reason to privilege either the categorical or the numeric quasi-identifier, and it allows suppression and microaggregation to act with comparable intensity on both attributes rather than concentrating distortion on one of them [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref8">8</xref>]. The geometric mean is the aggregation function used by ARX to combine the normalized per-attribute information losses into a single utility score, and it penalizes uneven loss across attributes more strongly than an arithmetic mean would, which is appropriate when no single attribute should be sacrificed to preserve another [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. The permissive suppression limit was chosen specifically to test whether a high admissible suppression budget would translate into aggressive data deletion. The optimization function was active throughout, and the result is informative in itself. Despite a 50% admissible suppression budget, the optimizer satisfied k-anonymity by masking fewer than 1% of encounters at k=5 and fewer than 3% at k=15 (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendices 2</xref><xref ref-type="supplementary-material" rid="app3"/>-<xref ref-type="supplementary-material" rid="app4">4</xref>), which shows that the observed distortion is not an artifact of brute-force deletion under a loose limit but a structural property of the suppression-plus-microaggregation configuration. The search strategy was the deterministic optimal solution, so the search space contained exactly one transformation per k level; because the ICD hierarchy height was fixed at 1 and no other generalization degrees of freedom existed, there were no alternative thresholds for the tool to optimize across, and reporting a single deterministic solution removes a source of run-to-run variability from the comparison.</p><p>However, the equivalence classes that failed to reach the k threshold were resolved exclusively through suppression. In ARX, suppression of this kind does not delete records [<xref ref-type="bibr" rid="ref30">30</xref>]. The tool replaces the affected quasi-identifier values with a missing token and continues to count and aggregate over the dataset until the k threshold is satisfied. For each encounter that could not be placed in a sufficiently large equivalence class, the primary ICD and the LOS were masked jointly, which confirms record-level rather than independent per-attribute masking. For LOS, microaggregation grouped records into clusters and replaced individual values with the cluster centroid. The geometric mean was used as the microaggregation function so that the imputed centroid is consistent with the logarithmic scale on which the length-of-stay outcome is subsequently modeled. For the strictly positive length-of-stay values in this cohort, the geometric mean equals the exponential of the mean of the log-transformed values, which may reduce the positive bias that arithmetic-mean substitution would introduce under a concave transformation. ARX preserves the original data type of the microaggregated attribute, so the geometric-mean centroids were returned as integer-valued length-of-stay days consistent with the discrete scale of the input variable rather than as continuous floating-point values, and all subsequent statistical tests and model fits were performed on this integer-preserving output.</p></sec><sec id="s2-6"><title>Design Note on Analytical Circularity and Study Scope</title><p>In this study, LOS serves a dual role. It is designated as a quasi-identifier subject to microaggregation by ARX, and it is simultaneously the response variable in the inferential reproducibility analysis (RQ2). This configuration is a deliberate worst-case evaluation design and is central to the aim of the study. Because microaggregation directly transforms the outcome variable, any observed inferential degradation represents an upper bound on the distortion that would occur when the analytical target is not itself part of the quasi-identifier set. We retained this design intentionally, because the purpose of the study is methodological. We use ICD code and LOS to demonstrate that distributional and inferential distortions can occur under self-configured anonymity and utility settings without the anonymization tool reporting them and to argue that effect-assessment findings of the kind reported here should accompany any analysis performed on anonymized data elements. Introducing an additional clinical outcome that is independent of the quasi-identifier set would weaken the worst-case demonstration and is not necessary for this objective, because the contribution of the study is the transparent quantification and reporting of the transformation footprint, not a substantive clinical claim about LOS. Analysts should interpret the inferential results accordingly. Studies in which the analytical outcome variable is not subject to anonymization transformation may experience substantially less inferential disruption than reported here, and the magnitude reported here should therefore be read as a ceiling.</p><p>The selection of exactly 2 quasi-identifiers is likewise a deliberate mechanism-focused design rather than an attempt to model a comprehensive real-world data release. ICD codes and LOS represent the 2 most frequent clinical data types requiring anonymization in hospital settings, namely, a high-dimensional categorical field and a right-skewed numerical outcome. Comparable variable pairings have been examined by Ohno-Machado et al [<xref ref-type="bibr" rid="ref12">12</xref>], who evaluated suppression effects on diagnostic codes and outcomes; by Pilgram et al [<xref ref-type="bibr" rid="ref13">13</xref>], who assessed anonymization costs using diagnosis and laboratory data; and by Johann et al [<xref ref-type="bibr" rid="ref14">14</xref>], who quantified utility loss for clinical scoring variables. By restricting the quasi-identifier set to these 2 elements, the study isolates the interplay between suppression and microaggregation while reducing additional sources of distortion from broader anonymization settings. Variables such as age, sex, and admission date were available in the source warehouse but intentionally excluded, because their inclusion would confound mechanism-specific effects with additional generalization or suppression pathways outside the scope of the RQs. The consequence for external validity is addressed explicitly in the Limitations and Future Work section.</p></sec><sec id="s2-7"><title>Distributional Impact Assessment</title><p>A purpose-built R function, quantify_anonymization_impact(), was developed to systematically compare original and anonymized datasets [<xref ref-type="bibr" rid="ref31">31</xref>]. The function accepts 2 data frames and optionally 2 column vectors for targeted ad-hoc comparison.</p><sec id="s2-7-1"><title>Numeric Distributional Analysis</title><p>For each shared numeric column, the function computes the 2-sample KS test. The KS <italic>D</italic> statistic evaluates the maximum vertical distance between empirical cumulative distribution functions [<xref ref-type="bibr" rid="ref32">32</xref>], <italic>D</italic>=sup|<italic>F</italic>0(<italic>x</italic>)&#x2013;<italic>F</italic>1(<italic>x</italic>)|. In this study, KS <italic>D</italic> is treated exclusively as a descriptive divergence measure quantifying the magnitude of distributional shift. It does not serve as a calibrated clinical effect size. With sample sizes exceeding 700,000, any nontrivial distributional deviation yields a <italic>D</italic> value that is statistically significant, which renders hypothesis-test interpretation uninformative [<xref ref-type="bibr" rid="ref33">33</xref>-<xref ref-type="bibr" rid="ref36">36</xref>]. Instead, <italic>D</italic> anchors a relative comparison across k levels, where a stable <italic>D</italic> indicates that incremental k increases do not add distributional perturbation beyond the initial transformation step. The scaled statistic <italic>z</italic>=sqrt(neff)*<italic>D</italic> is reported alongside <italic>D</italic> to confirm that the divergence is both large and stable in magnitude across levels, and not as a threshold criterion.</p><p>To complement the KS <italic>D</italic>, 2 additional distributional descriptors are reported. Quantile shifts capture changes at the median, providing a clinically interpretable reference point. The IQR change documents the compression of central dispersion. Together, these metrics ground the variance-collapse argument in quantities directly meaningful to hospital resource planning and clinical research [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. The KS and the complementary quantitative metrics were applied to hospital LOS, the single numeric data element in the study data.</p></sec><sec id="s2-7-2"><title>Categorical Analysis</title><p>For each shared categorical column, 3 metrics are computed. The Jaccard similarity coefficient <italic>J</italic>(A,B)=|A intersect B|/|A union B| measures label-space erosion, quantifying whether codes survive anonymization [<xref ref-type="bibr" rid="ref37">37</xref>,<xref ref-type="bibr" rid="ref38">38</xref>]. Cramer <italic>V</italic>=sqrt(<italic>&#x03C7;</italic><sup>2</sup>/(n*min(<italic>r</italic>&#x2013;1, <italic>c</italic>&#x2013;1))) quantifies the magnitude of frequency redistribution among retained shared categories [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. The chi-square test underlying Cramer <italic>V</italic> is computed over shared levels only [<xref ref-type="bibr" rid="ref41">41</xref>], so the effect size reflects distributional shift rather than conflating it with vocabulary loss. The 2 metrics address complementary facets of categorical degradation. Jaccard captures structural completeness, while Cramer <italic>V</italic> captures distributional fidelity within the surviving code set. They are not redundant.</p></sec><sec id="s2-7-3"><title>Composite Verdict</title><p>The composite verdict is an illustrative heuristic layer designed to make the detailed metrics interpretable for decision-makers and applied researchers. Each comparison yields a traffic-light verdict (green, yellow, or red) based on fixed thresholds for the constituent metrics. A green verdict indicates that the anonymized variable remains suitable for analyses that depend on distributional properties, including variance-dependent methods and prediction models. A yellow verdict signals that comparative or ordinal analyses remain feasible, but analyses dependent on distributional shape or absolute magnitudes require caution. A red verdict warns that anonymization-induced distortion may alter substantive conclusions.</p><p>Numeric variables are evaluated in 3 dimensions. For distributional divergence, the KS <italic>D</italic> statistic serves as a summary index of overall distributional shift, with values below 0.05 indicating negligible shift, values between 0.05 and 0.10 indicating moderate shift, and values at or above 0.10 indicating substantial shift. These bands are pragmatic screening thresholds calibrated to the present evaluation context, not universal effect-size benchmarks, and the KS <italic>D</italic> is read jointly with the complementary descriptors of median shift and IQR change. For location shift, the absolute mean shift divided by the original SD produces a dimensionless ratio, classified as tiny at or below 0.10, moderate up to 0.30, and large above 0.30. For spread stability, the SD shift relative to the baseline SD captures whether dispersion is preserved, with deviations within &#x00B1;10% considered stable and larger deviations indicating compression or inflation. A green verdict requires all 3 conditions jointly, namely, <italic>D</italic> below 0.05, normalized mean shift at or below 0.10, and SD change within &#x00B1;10%. Yellow requires <italic>D</italic> below 0.10 combined with a normalized mean shift at or below 0.30. Every remaining combination triggers red. Categorical variables are evaluated on 3 parallel dimensions. A green verdict requires Cramer <italic>V</italic> below 0.10, Jaccard overlap at or above 0.80, and missing-data drift at or below 5 percentage points. Yellow permits Cramer <italic>V</italic> below 0.30, Jaccard at or above 0.60, and drift at or below 10 percentage points. All other combinations produce red.</p></sec><sec id="s2-7-4"><title>Threshold Justification</title><p>The normalized mean shift thresholds (0.10 and 0.30) align with the standardized effect-size conventions of Cohen for small and medium effects [<xref ref-type="bibr" rid="ref40">40</xref>]. The SD stability window of &#x00B1;10% follows analytical precision standards applied in laboratory medicine and biostatistics [<xref ref-type="bibr" rid="ref42">42</xref>]. The categorical thresholds for Cramer <italic>V</italic> (0.10 and 0.30) correspond to the benchmarks of Cohen for small and medium association effects [<xref ref-type="bibr" rid="ref40">40</xref>]. The Jaccard thresholds (0.80 and 0.60) were informed by Haizoune et al [<xref ref-type="bibr" rid="ref43">43</xref>], who applied comparable composite metrics for privacy-utility assessment. The KS <italic>D</italic> screening bands do not derive from a single established convention. They were set to distinguish negligible, moderate, and substantial distributional displacement within this evaluation approach, and their primary function is to enable consistent relative comparison across k levels rather than to serve as absolute cutoffs.</p></sec></sec><sec id="s2-8"><title>LMM Specification</title><sec id="s2-8-1"><title>Overview</title><p>An LMM was selected because the analytical goal of RQ2 is not optimal prediction of LOS but a stable, comparable decomposition of LOS variance into a diagnosis-level component and the remaining variation, evaluated under an identical model form across the original and the 3 anonymized datasets. Three considerations motivate this choice. First, the ICD code grouped to 3 characters (ICD-3) is a high-cardinality categorical variable with around 1000 levels [<xref ref-type="bibr" rid="ref44">44</xref>], which is impractical to estimate as a fixed-effect factor and natural to treat as a grouping variable. Second, a random-intercept specification treats ICD-3 categories as drawn from a common distribution and applies shrinkage, which stabilizes the diagnosis-level estimates that RQ2 compares before and after anonymization [<xref ref-type="bibr" rid="ref45">45</xref>]. Third, and most importantly for this study, the random-effect variance components and the diagnosis-level best linear unbiased predictions (BLUPs) provide exactly the quantities needed to investigate whether the diagnosis-to-LOS signal is reproducible after anonymization, namely, the intraclass correlation coefficient (ICC) and the per-diagnosis effect estimates. The LMM is therefore the measurement instrument for reproducibility, not a predictive model whose fit is of interest in its own right.</p><p>The model was specified as a linear mixed-effects model with crossed random intercepts for ICD-3 category and patient. To address nonindependence of repeated encounters from the same patient, a patient-level random intercept was included alongside the ICD-3 random intercept, while admission year was included as a fixed categorical effect. The outcome was transformed as log(LOS+1), rather than modeled on the raw LOS scale, to stabilize variance and reduce right-skewness in the length-of-stay distribution. Although same-day encounters had already been assigned a LOS value of 1 during data preparation, the same log(LOS+1) transformation was applied consistently across the original and anonymized datasets. Thus, the fitted model was log(LOS+1)=<italic>b</italic>0+<italic>&#x03B3;t</italic>+<italic>uj</italic>+<italic>vp</italic>+<italic>eijpt</italic>, where <italic>b</italic>0 is the fixed intercept, <italic>&#x03B3;t</italic> denotes the fixed categorical effect of admission year <italic>t</italic>, <italic>uj</italic>~<italic>N</italic>(0, &#x03C3;&#x00B2;ICD-3) is the random intercept for ICD-3 category <italic>j</italic>, <italic>vp</italic>~<italic>N</italic>(0, &#x03C3;&#x00B2;patient) is the random intercept for patient <italic>p</italic>, and <italic>eijpt</italic>~<italic>N</italic>(0, &#x03C3;&#x00B2;residual) is the within-encounter residual. Admission year was also retained for year-specific summaries of cell-level ICD and LOS suppression. Estimation used restricted maximum likelihood with the <italic>lme4</italic> package [<xref ref-type="bibr" rid="ref46">46</xref>,<xref ref-type="bibr" rid="ref47">47</xref>]. The values of Akaike information criterion (AIC) and Bayesian information criterion were extracted from the restricted maximum likelihood&#x2013;fitted models.</p></sec><sec id="s2-8-2"><title>Note on Negative Binomial Regression</title><p>A negative binomial generalized linear model was initially considered as an alternative. Fitting MASS::glm.nb() with ICD-3 as a high-cardinality fixed effect across datasets of approximately 700,000 observations did not converge for any dataset, and the dispersion parameter, AIC, and coefficient estimates could not be obtained. This nonconvergence underscores the value of the random-effects approach, which remains computationally tractable regardless of predictor dimensionality.</p></sec></sec><sec id="s2-9"><title>Reproducibility Assessment</title><p>Inferential reproducibility was operationalized as the degree to which the diagnosis-level signal recovered from anonymized data agrees with the signal recovered from the original data, assessed across 5 complementary quantities that separate variance structure, rank agreement, and magnitude agreement. First, the ICC=&#x03C3;2_icd3/(&#x03C3;2_icd3+&#x03C3;2_patient+&#x03C3;2_residual) quantifies the proportion of total log-LOS variance attributable to diagnosis-group membership [<xref ref-type="bibr" rid="ref48">48</xref>], and the Nakagawa-Schielzeth <italic>R</italic><sup>2</sup> provides a complementary variance decomposition [<xref ref-type="bibr" rid="ref49">49</xref>]. Second, rank agreement of the diagnosis-level BLUPs between the original and each anonymized model is assessed with the Spearman rank correlation &#x03C1;_S [<xref ref-type="bibr" rid="ref50">50</xref>], which answers whether diagnoses keep their relative position in the LOS hierarchy. Third, magnitude agreement is assessed with the Lin concordance correlation coefficient (CCC) [<xref ref-type="bibr" rid="ref51">51</xref>], which evaluates rank preservation and magnitude agreement simultaneously by measuring how closely paired BLUP estimates fall on the identity line, and which decomposes into a precision component (Pearson <italic>r</italic>) and an accuracy component (bias-correction factor <italic>C</italic>b) [<xref ref-type="bibr" rid="ref51">51</xref>], with interpretive strength-of-agreement bands proposed by McBride [<xref ref-type="bibr" rid="ref52">52</xref>]. Reporting both &#x03C1;_S and CCC follows the principle that a rank correlation can remain high even when absolute magnitudes diverge [<xref ref-type="bibr" rid="ref53">53</xref>]. Fourth, the fraction of ICD-3 categories whose BLUP changes sign between the original and anonymized model is reported, since a sign reversal indicates a qualitative reinterpretation of the diagnosis-to-LOS relationship, and the median absolute deviation of BLUP pairs provides a magnitude-focused dispersion measure that is resistant to outliers. Fifth, model fit and prediction accuracy are tracked through AIC, Bayesian information criterion, log-likelihood, root mean square error (RMSE), and mean absolute error. To quantify the precision of the 2 agreement statistics, 95% CIs were computed for the Spearman correlation and for the Lin CCC. All analyses were conducted in R (version 4.4.0; R Foundation for Statistical Computing).</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Cohort Description</title><p>The cohort extraction yielded 719,387 inpatient encounters belonging to 370,548 distinct patients, spanning January 2010 through September 2024. LOS ranged from 1 to 557 days, with a mean of 6.52 (SD 9.74) days, and a median of 4 (IQR 2-7) days. Because the cohort was restricted to encounters with valid admission and discharge dates, every encounter had a defined LOS, and no encounter was excluded for missing LOS. The 5 most prevalent primary ICD-10-GM codes were Z38.0 (single live birth, 2.4%), G47.31 (obstructive sleep apnea), C61 (prostate carcinoma), S06.0 (concussion), and I63.4 (cerebral infarction). No single diagnosis exceeded 3%, confirming a diffuse diagnostic landscape typical of tertiary university hospitals. The original dataset contained 7715 distinct full ICD-10-GM codes, mapping to 1404 unique 3-character prefix categories (ICD-3), which served as the diagnosis grouping variable for all mixed-model analyses.</p></sec><sec id="s3-2"><title>Anonymization Footprint</title><p>ARX satisfied the anonymity criterion at every level with a deterministic search space of size 1 (<xref ref-type="supplementary-material" rid="app2">Multimedia Appendices 2</xref><xref ref-type="supplementary-material" rid="app3"/>-<xref ref-type="supplementary-material" rid="app4">4</xref>) and reported internal information-loss values (geometric mean) of 0.074 at k=5, 0.084 at k=10, and 0.093 at k=15. Consistent with the mechanism described in the Methods section, ARX did not delete any encounters. The number of rows was identical across the original and all 3 anonymized datasets, and the privacy guarantee was achieved by masking quasi-identifier cells with a missing token (data_point_removal.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). The proportion of encounters carrying at least 1 masked quasi-identifier cell rose from 0.77% at k=5 (5564 encounters) to 1.71% at k=10 (12,315 encounters) and 2.62% at k=15 (18,858 encounters), and in every masked encounter, the primary ICD and the LOS were masked jointly, confirming record-level rather than independent per-attribute masking (<xref ref-type="table" rid="table2">Table 2</xref>). The masking was therefore small in volume relative to the cohort, even though the admissible suppression budget was 50%, which indicates that the optimizer did not resort to aggressive deletion to reach the privacy threshold.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Anonymization summary and suppression documentation across k-anonymity thresholds<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Original</td><td align="left" valign="bottom">k=5</td><td align="left" valign="bottom">k=10</td><td align="left" valign="bottom">k=15</td></tr></thead><tbody><tr><td align="left" valign="top">Total records (rows), n</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td></tr><tr><td align="left" valign="top">Encounters removed (rows deleted), n</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td><td align="left" valign="top">0</td></tr><tr><td align="left" valign="top">Encounters with &#x2265;1 masked quasi-identifier cell, n (%)</td><td align="left" valign="top">0 (0)</td><td align="left" valign="top">5564 (0.77)</td><td align="left" valign="top">12,315 (1.71)</td><td align="left" valign="top">18,858 (2.62)</td></tr><tr><td align="left" valign="top">Encounters with complete ICD<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup> and LOS<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>, n (%)</td><td align="left" valign="top">719,387 (100)</td><td align="left" valign="top">713,823 (99.23)</td><td align="left" valign="top">707,072 (98.29)</td><td align="left" valign="top">700,529 (97.38)</td></tr><tr><td align="left" valign="top">Distinct full ICD codes, n</td><td align="left" valign="top">7715</td><td align="left" valign="top">4814</td><td align="left" valign="top">3803</td><td align="left" valign="top">3256</td></tr><tr><td align="left" valign="top">Information loss (geometric mean)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.074</td><td align="left" valign="top">0.084</td><td align="left" valign="top">0.093</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>k denotes the minimum equivalence-class size required by the k-anonymity model. Suppression in ARX masks quasi-identifier cells with a missing token rather than deleting rows. A total of 719,387 inpatient encounters from University Hospital Mannheim (2010&#x2010;2024).</p></fn><fn id="table2fn2"><p><sup>b</sup>ICD: International Classification of Diseases.</p></fn><fn id="table2fn3"><p><sup>c</sup>LOS: length of stay.</p></fn><fn id="table2fn4"><p><sup>d</sup>Not available.</p></fn></table-wrap-foot></table-wrap><p>The masking was distributed almost uniformly across the 15 admission years rather than concentrated in particular periods (suppression_by_year.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Year-specific masking rates ranged narrowly from 0.70% to 0.87% at k=5, from 1.58% to 1.92% at k=10, and from 2.36% to 2.88% at k=15, with no monotonic trend across the study period. Unadjusted temporal drift in the suppression pattern is therefore minimal, and the diagnosis-level comparisons that follow are not driven by a particular admission year.</p></sec><sec id="s3-3"><title>Distributional Fidelity (RQ1)</title><sec id="s3-3-1"><title>Length of Stay</title><p>The KS <italic>D</italic> statistic, interpreted as a descriptive divergence measure, registered 0.147 at k=5, 0.147 at k=10, and 0.147 at k=15 (<xref ref-type="table" rid="table3">Table 3</xref>). The corresponding scaled statistic <italic>z</italic> was 87.80, 87.81, and 87.59, confirming that the divergence is both large and, importantly, near-constant in magnitude across levels. This near-identity of <italic>D</italic> and <italic>z</italic> across the 3 thresholds indicates that the dominant distributional shift occurs at the first anonymization step, with negligible incremental perturbation at higher k.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Consolidated distributional and inferential metrics across anonymization levels (N=719,387 encounters, 370,548 patients).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom"><italic>D</italic>0<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom"><italic>D</italic>5<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (k=5)</td><td align="left" valign="bottom"><italic>D</italic>10<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (k=10)</td><td align="left" valign="bottom"><italic>D</italic>15<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup> (k=15)</td></tr></thead><tbody><tr><td align="left" valign="top">Records (rows), n</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td><td align="left" valign="top">719,387</td></tr><tr><td align="left" valign="top">Encounters with complete ICD<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup> and LOS<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup>, n</td><td align="left" valign="top">719,387</td><td align="left" valign="top">713,823</td><td align="left" valign="top">707,072</td><td align="left" valign="top">700,529</td></tr><tr><td align="left" valign="top">ICD-3<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup> groups, n</td><td align="left" valign="top">1404</td><td align="left" valign="top">1170</td><td align="left" valign="top">1059</td><td align="left" valign="top">979</td></tr><tr><td align="left" valign="top">Groups not estimable versus <italic>D</italic>0</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="left" valign="top">234</td><td align="left" valign="top">345</td><td align="left" valign="top">425</td></tr><tr><td align="left" valign="top">LOS KS <italic>D</italic><sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup> (descriptive)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.147</td><td align="left" valign="top">0.147</td><td align="left" valign="top">0.147</td></tr><tr><td align="left" valign="top">LOS KS <italic>z</italic> (scaled)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">87.80</td><td align="left" valign="top">87.81</td><td align="left" valign="top">87.59</td></tr><tr><td align="left" valign="top">LOS median (IQR) (days)</td><td align="left" valign="top">4 (2-7)</td><td align="left" valign="top">4 (3-6)</td><td align="left" valign="top">4 (3-6)</td><td align="left" valign="top">4 (3-6)</td></tr><tr><td align="left" valign="top">LOS mean shift (SD) (days)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">&#x2212;1.81 (&#x2212;6.45)</td><td align="left" valign="top">&#x2212;1.81 (&#x2212;6.47)</td><td align="left" valign="top">&#x2212;1.82 (&#x2212;6.48)</td></tr><tr><td align="left" valign="top">LOS verdict</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Red</td><td align="left" valign="top">Red</td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">ICD Jaccard</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.624</td><td align="left" valign="top">0.493</td><td align="left" valign="top">0.421</td></tr><tr><td align="left" valign="top">ICD Cramer <italic>V</italic></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.062</td><td align="left" valign="top">0.093</td><td align="left" valign="top">0.114</td></tr><tr><td align="left" valign="top">ICD verdict</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Yellow</td><td align="left" valign="top">Red</td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">&#x03C3;2_icd3<sup><xref ref-type="table-fn" rid="table3fn8">h</xref></sup></td><td align="left" valign="top">0.200</td><td align="left" valign="top">0.231</td><td align="left" valign="top">0.228</td><td align="left" valign="top">0.234</td></tr><tr><td align="left" valign="top">&#x03C3;2_patient<sup><xref ref-type="table-fn" rid="table3fn9">i</xref></sup></td><td align="left" valign="top">0.081</td><td align="left" valign="top">0.010</td><td align="left" valign="top">0.010</td><td align="left" valign="top">0.010</td></tr><tr><td align="left" valign="top">&#x03C3;2_residual<sup><xref ref-type="table-fn" rid="table3fn10">j</xref></sup></td><td align="left" valign="top">0.398</td><td align="left" valign="top">0.038</td><td align="left" valign="top">0.036</td><td align="left" valign="top">0.036</td></tr><tr><td align="left" valign="top">ICC<sup><xref ref-type="table-fn" rid="table3fn11">k</xref></sup> (diagnosis)</td><td align="left" valign="top">0.294</td><td align="left" valign="top">0.827</td><td align="left" valign="top">0.831</td><td align="left" valign="top">0.837</td></tr><tr><td align="left" valign="top">ICC (patient)</td><td align="left" valign="top">0.120</td><td align="left" valign="top">0.037</td><td align="left" valign="top">0.036</td><td align="left" valign="top">0.035</td></tr><tr><td align="left" valign="top">ICC (diagnosis+patient)</td><td align="left" valign="top">0.414</td><td align="left" valign="top">0.864</td><td align="left" valign="top">0.867</td><td align="left" valign="top">0.873</td></tr><tr><td align="left" valign="top">Intercept b0 (SE)</td><td align="left" valign="top">1.634 (0.013)</td><td align="left" valign="top">1.596 (0.014)</td><td align="left" valign="top">1.598 (0.015)</td><td align="left" valign="top">1.597 (0.016)</td></tr><tr><td align="left" valign="top">BLUP<sup><xref ref-type="table-fn" rid="table3fn12">l</xref></sup> &#x03C1;_S<sup><xref ref-type="table-fn" rid="table3fn13">m</xref></sup> versus <italic>D</italic>0</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.970</td><td align="left" valign="top">0.966</td><td align="left" valign="top">0.964</td></tr><tr><td align="left" valign="top">BLUP &#x03C1;_S 95% CI</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.963&#x2010;0.975</td><td align="left" valign="top">0.959&#x2010;0.973</td><td align="left" valign="top">0.956&#x2010;0.971</td></tr><tr><td align="left" valign="top">BLUP Lin CCC<sup><xref ref-type="table-fn" rid="table3fn14">n</xref></sup></td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.966</td><td align="left" valign="top">0.965</td><td align="left" valign="top">0.959</td></tr><tr><td align="left" valign="top">BLUP CCC 95% CI</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.960&#x2010;0.971</td><td align="left" valign="top">0.959&#x2010;0.970</td><td align="left" valign="top">0.948&#x2010;0.967</td></tr><tr><td align="left" valign="top">BLUP Pearson <italic>r</italic> (CCC precision)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.971</td><td align="left" valign="top">0.969</td><td align="left" valign="top">0.964</td></tr><tr><td align="left" valign="top">BLUP CCC bias-correction <italic>C</italic>b (accuracy)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.995</td><td align="left" valign="top">0.995</td><td align="left" valign="top">0.994</td></tr><tr><td align="left" valign="top">BLUP sign reversals (%)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">6.8</td><td align="left" valign="top">6.7</td><td align="left" valign="top">7.3</td></tr><tr><td align="left" valign="top">Median BLUP ratio (<italic>D</italic>k/<italic>D</italic>0)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">1.049</td><td align="left" valign="top">1.044</td><td align="left" valign="top">1.049</td></tr><tr><td align="left" valign="top">BLUP pair MAD<sup><xref ref-type="table-fn" rid="table3fn15">o</xref></sup> (<italic>D</italic>k<italic>&#x2212;D</italic>0)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.077</td><td align="left" valign="top">0.082</td><td align="left" valign="top">0.084</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup><italic>D</italic>0: original dataset.</p></fn><fn id="table3fn2"><p><sup>b</sup><italic>D</italic>5/<italic>D</italic>10/<italic>D</italic>15: k-anonymized variants.</p></fn><fn id="table3fn3"><p><sup>c</sup>ICD: International Classification of Diseases.</p></fn><fn id="table3fn4"><p><sup>d</sup>LOS: length of stay.</p></fn><fn id="table3fn5"><p><sup>e</sup>ICD-3: three-character ICD category.</p></fn><fn id="table3fn6"><p><sup>f</sup>Not available.</p></fn><fn id="table3fn7"><p><sup>g</sup>KS <italic>D</italic>: Kolmogorov-Smirnov <italic>D</italic> statistic (descriptive divergence).</p></fn><fn id="table3fn8"><p><sup>h</sup>&#x03C3;2_icd3: between-diagnosis variance.</p></fn><fn id="table3fn9"><p><sup>i</sup>&#x03C3;2_patient: between-patient variance.</p></fn><fn id="table3fn10"><p><sup>j</sup>&#x03C3;2_residual: within-encounter residual variance.</p></fn><fn id="table3fn11"><p><sup>k</sup>ICC: intraclass correlation coefficient. All ICC values were computed from the full-precision unrounded variance components.</p></fn><fn id="table3fn12"><p><sup>l</sup>BLUP: best linear unbiased prediction.</p></fn><fn id="table3fn13"><p><sup>m</sup>&#x03C1;_S: Spearman rank correlation.</p></fn><fn id="table3fn14"><p><sup>n</sup>CCC: concordance correlation coefficient.</p></fn><fn id="table3fn15"><p><sup>o</sup>MAD: median absolute deviation.</p></fn></table-wrap-foot></table-wrap><p>Complementary descriptors characterize the shift. At the distribution level, microaggregation left the median LOS unchanged, from 4 (IQR 2-7) days in the original data to 4 (IQR 3-6) days in every anonymized variant, and narrowed the IQR from 5 to 3 days. This distributional shift, driven by cluster-centroid substitution, is distinct from the cohort-level comparison of masked versus retained encounters reported below, where the median LOS is 4 days in both groups (masked: 4, IQR 2-8; retained: 4, IQR 2-7). The behavior of the statistical moments was asymmetric. The diagnosis-level mean of LOS shifted considerably (absolute mean shift of &#x2212;1.81 days), whereas the SD compressed by approximately 6.5 days at every threshold (mean shift about &#x2212;1.81 days and SD shift about &#x2212;6.45 days on the encounter scale; summary_metrics.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Microaggregation replaces within-cluster LOS values with cluster means, which attenuates the second moment far more than the first. An analyst relying on simple location summaries would therefore underestimate the change, while any analysis dependent on variance, CIs, or distributional shape would operate on a materially homogenized outcome. The empirical cumulative distribution function comparison showed that the 3 anonymized LOS curves overlap almost entirely and are displaced from the original curve, reinforcing the saturation pattern (<xref ref-type="fig" rid="figure1">Figure 1</xref>). Because suppression and microaggregation were applied jointly, the analysis does not isolate which subset of encounters contributed most strongly to the displacement.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Empirical cumulative distribution functions of hospital length of stay. Black: original dataset (<italic>D</italic>0). Colored: k-anonymized variants. Near-complete overlap among the k=5, k=10, and k=15 curves indicates that the distributional shift saturates at k=5. CDF: cumulative distribution function; LOS: length of stay.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e97661_fig01.png"/></fig></sec><sec id="s3-3-2"><title>ICD Codes</title><p>The Jaccard similarity coefficient declined monotonically from 0.624 at k=5 through 0.493 at k=10 to 0.421 at k=15, reflecting progressive vocabulary erosion. Cramer <italic>V</italic> increased from 0.062 to 0.093 and 0.114. At the strictest threshold, more than half of the original ICD-10-GM codes no longer appeared. Categorical degradation, unlike LOS distortion, exhibited a clear dose-response gradient (<xref ref-type="fig" rid="figure2">Figure 2</xref>).</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>ICD code quality, label preservation versus distributional shift. Left axis: Jaccard overlap ratio quantifying the fraction of original ICD-10-GM codes retained across thresholds. Right axis: Cramer <italic>V</italic> measuring the effect size of frequency redistribution among surviving codes. Horizontal dashed lines mark small (<italic>V</italic>=0.10) and medium (<italic>V</italic>=0.30) effect-size benchmarks. Jaccard captures progressive code-set attrition through suppression of rare diagnoses, while Cramer <italic>V</italic> captures the amplification of association strength as residual frequency mass concentrates on common codes. Both effects intensify monotonically with k. ICD: International Classification of Diseases; ICD-10-GM: International Classification of Diseases, 10th Revision, German Modification.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e97661_fig02.png"/></fig></sec></sec><sec id="s3-4"><title>Inferential Reproducibility (RQ2)</title><sec id="s3-4-1"><title>Variance Decomposition and ICC</title><p>The 3-level LMM fitted to the original data attributed 29.4% of log-LOS variance to ICD-3 category (ICC_icd3=0.294) and a further 12% to the patient level (ICC_patient=0.120), with diagnosis and patient together explaining 41.4% of the variance (<xref ref-type="table" rid="table3">Table 3</xref>). The between-diagnosis variance (&#x03C3;2_icd3) was 0.200, the between-patient variance (&#x03C3;2_patient) was 0.081, and the residual variance (&#x03C3;2_residual) was 0.398. After anonymization, the ICC attributable to diagnosis rose sharply to 0.827 at k=5, 0.831 at k=10, and 0.837 at k=15. This near-tripling must be interpreted with caution. It is a consequence of microaggregation-induced compression of the residual variance combined with selective masking of atypical records, not an indicator of improved diagnostic explanatory power. Between-diagnosis variance rose moderately from 0.200 to between 0.228 and 0.234, while residual variance collapsed roughly 10-fold from 0.398 to between 0.036 and 0.038, and between-patient variance fell from 0.081 to about 0.010. Because the ICC is a ratio of a between-group variance to the total variance, an operation that disproportionately reduces the denominator inflates the ICC mechanically [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref54">54</xref>-<xref ref-type="bibr" rid="ref57">57</xref>]. Elevated ICC values in anonymized data should therefore be read as a warning signal of compressed outcome variability, not as evidence that diagnosis categories explain clinical LOS variation more effectively. The number of ICD-3 groups represented in the model decreased from 1404 in the original data to 1170 at k=5, 1059 at k=10, and 979 at k=15, corresponding to 234, 345, and 425 categories no longer estimable after anonymization.</p></sec><sec id="s3-4-2"><title>Between-Diagnosis Variance</title><p>The moderate rise in between-diagnosis variance was analyzed (variance_inflation_interrogation.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Between-diagnosis variance increased by 15.5% at k=5, 14.1% at k=10, and 17% at k=15 relative to the original value. Over the same range, the number of ICD-3 groups fell from 1404 to 1170, 1059, and 979, and the median group size grew from 91 encounters to 148, 174, and 203. The mechanism is consistent across these quantities. Suppression preferentially removes small, low-frequency diagnosis groups, so the surviving groups are larger, more homogeneous internally, and more separated from one another after microaggregation has collapsed within-group variation. The between-group variance therefore inflates not because diagnoses became more informative, but because the surviving set of diagnoses is sparser and more widely spaced once the rare groups and the within-group spread have been removed.</p></sec><sec id="s3-4-3"><title>Model Diagnostics</title><p>Residual inspection on the log scale showed acceptable symmetry. No patterns threatened the group-level comparison (Fig_QQ_residuals_D0 through Fig_QQ_residuals_D15 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Quantile-quantile plots of the ICD-3 BLUPs confirmed approximate normality of the diagnosis random-intercept distribution. This held across all 4 datasets (Fig_QQ_BLUPs_D0 through Fig_QQ_BLUPs_D15 in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). The Shapiro-Wilk test on 5000 subsampled residuals (seed=42) indicated significant departures from normality in every dataset. This is expected at N exceeding 700,000. At this scale, the central limit theorem ensures robust variance-component estimation regardless of moderate distributional deviation [<xref ref-type="bibr" rid="ref44">44</xref>]. All 4 models converged without warnings or boundary estimates (final_analysis.R in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>).</p></sec><sec id="s3-4-4"><title>BLUP Concordance, Rank Order, and Magnitude</title><p>Spearman &#x03C1; confirmed preserved rank ordering of diagnosis-level effects, at 0.970 (95% CI 0.963-0.975) at k=5, 0.966 (95% CI 0.959-0.973) at k=10, and 0.964 (95% CI 0.956-0.971) at k=15, all with <italic>P</italic> values below machine precision (<xref ref-type="fig" rid="figure3">Figure 3</xref> and BLUP_concordance.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). A researcher asking which diagnoses carry the longest stays would, in this case study, arrive at concordant rankings from anonymized data. Lin CCC, which integrates precision and accuracy, was also high, at 0.966 (95% CI 0.960-0.971) at k=5, 0.965 (95% CI 0.959-0.970) at k=10, and 0.959 (95% CI 0.948-0.967) at k=15, with Pearson <italic>r</italic> between 0.964 and 0.971. The CIs are narrow, which reflects the large number of shared diagnosis groups on which the agreement statistics are computed (1,170, 1,059, and 979 shared groups). The median BLUP attenuation ratio was close to unity (about 1.04 to 1.05), indicating that the typical diagnosis-level effect was reproduced with little systematic shrinkage or amplification. Decomposing the Lin CCC into its precision and accuracy components confirmed that the high agreement was not a coincidence of offsetting errors. The bias-correction factor <italic>C</italic>b, which isolates the accuracy component, was 0.995, 0.995, and 0.994 across the 3 levels, so the paired estimates lay almost exactly on the identity line rather than on a shifted or rescaled line, and the CCC was therefore driven by precision (Pearson <italic>r</italic>) rather than by chance alignment. The median absolute deviation of the BLUP pairs, a magnitude-focused dispersion measure resistant to outliers, was small and rose only gradually with k, at 0.077, 0.082, and 0.084 on the log scale.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>BLUP concordance scatter plots, original versus anonymized data. Each point is one shared ICD-3 category. The Spearman rank correlation is annotated per panel, and the diagonal marks perfect concordance. BLUP: best linear unbiased prediction; CCC: concordance correlation coefficient; ICD-3: three-character ICD category.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e97661_fig03.png"/></fig></sec><sec id="s3-4-5"><title>Extreme Individual Distortions and Sign Reversals</title><p>High aggregate concordance coexisted with extreme distortion of a minority of individual diagnoses, which a single correlation coefficient would conceal (BLUP_extreme_distortion.csv and Fig_extreme_distortion in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). The diagnoses affected most severely were, almost without exception, those whose original effect was already close to 0, where a small absolute change produces a large relative change or a sign flip. For example, prostate carcinoma (C61) had an original BLUP of only 0.004 and an anonymized BLUP of &#x2212;0.218 at k=5, which registers as a sign reversal and a nominal ratio far outside the plausible range; nasal-cavity and middle-ear cancers (C30 and C31) behaved similarly. By contrast, diagnoses with substantial original effects were reproduced closely, for example, A02 was reproduced at about 64% of its original value and Z38 at about 77%. Across the shared groups, BLUP sign reversals occurred in approximately 7% of ICD-3 categories at each k level (k=5: 80/1170, k=10: 71/1059, and k=15: 71/979). Moreover, the reversed categories were overwhelmingly those with original effects near 0 on the log scale, where stochastic fluctuation dominates the estimate. This indicates that rank ordering and aggregate magnitude are preserved, while the effect estimate for any specific low-signal diagnosis can be unreliable after anonymization. Meta-analyses pooling anonymized and original estimates for such diagnoses could therefore produce biased summaries despite concordant overall ranking.</p></sec><sec id="s3-4-6"><title>Prediction Accuracy</title><p>As a secondary diagnostic of the variance-compression effect, prediction accuracy was computed across datasets (<xref ref-type="table" rid="table4">Table 4</xref>). RMSE on the day scale fell from 8.57 in the original data to about 1.58 at k=5 and 1.54 at k=15, and the AIC moved from 1,496,114 in the original data to large negative values (about &#x2212;161,308 at k=5 to &#x2212;203,119 at k=15). This apparent improvement does not reflect superior model quality. It is a direct consequence of fitting the model to a variance-depleted response, where the outcome has been compressed toward cluster means. The marginal <italic>R</italic><sup>2</sup>, which captures only the variance explained by the admission-year fixed effect, was near 0 in every dataset (0.005 in the original data and 0.000 after anonymization), confirming that admission year carried almost no explanatory weight and that the conditional <italic>R</italic><sup>2</sup> is driven entirely by the random effects. The rise in conditional <italic>R</italic><sup>2</sup> from 0.417 to above 0.86 therefore tracks the same residual-variance compression seen in the ICC rather than any genuine gain in explanatory power.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Model fit and prediction accuracy across anonymization levels<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup>.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom"><italic>D</italic>0</td><td align="left" valign="bottom"><italic>D</italic>5</td><td align="left" valign="bottom"><italic>D</italic>10</td><td align="left" valign="bottom"><italic>D</italic>15</td></tr></thead><tbody><tr><td align="left" valign="top">AIC<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top">1,496,114</td><td align="left" valign="top">&#x2013;161,308</td><td align="left" valign="top">&#x2013;188,380</td><td align="left" valign="top">&#x2013;203,119</td></tr><tr><td align="left" valign="top">BIC<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">1,496,320</td><td align="left" valign="top">&#x2013;161,101</td><td align="left" valign="top">&#x2013;188,173</td><td align="left" valign="top">&#x2013;202,912</td></tr><tr><td align="left" valign="top">Log-likelihood</td><td align="left" valign="top">&#x2013;748,039</td><td align="left" valign="top">80,672</td><td align="left" valign="top">94,208</td><td align="left" valign="top">101,577</td></tr><tr><td align="left" valign="top">RMSE<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup> (log)</td><td align="left" valign="top">0.589</td><td align="left" valign="top">0.179</td><td align="left" valign="top">0.176</td><td align="left" valign="top">0.173</td></tr><tr><td align="left" valign="top">MAE<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup> (log)</td><td align="left" valign="top">0.436</td><td align="left" valign="top">0.108</td><td align="left" valign="top">0.105</td><td align="left" valign="top">0.103</td></tr><tr><td align="left" valign="top">RMSE (days)</td><td align="left" valign="top">8.571</td><td align="left" valign="top">1.581</td><td align="left" valign="top">1.549</td><td align="left" valign="top">1.535</td></tr><tr><td align="left" valign="top">MAE (days)</td><td align="left" valign="top">3.516</td><td align="left" valign="top">0.656</td><td align="left" valign="top">0.639</td><td align="left" valign="top">0.628</td></tr><tr><td align="left" valign="top">Marginal <italic>R</italic><sup>2</sup></td><td align="left" valign="top">0.005</td><td align="left" valign="top">0.000</td><td align="left" valign="top">0.000</td><td align="left" valign="top">0.000</td></tr><tr><td align="left" valign="top">Conditional <italic>R</italic><sup>2</sup></td><td align="left" valign="top">0.417</td><td align="left" valign="top">0.864</td><td align="left" valign="top">0.867</td><td align="left" valign="top">0.873</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Negative AIC values and reduced RMSE in anonymized datasets reflect a variance-depleted response, not improved model quality (see the Discussion section). Marginal <italic>R</italic><sup>2</sup> reflects variance explained by the fixed effect (admission year) alone; conditional <italic>R</italic><sup>2</sup> additionally includes the diagnosis and patient random effects (Nakagawa-Schielzeth).</p></fn><fn id="table4fn2"><p><sup>b</sup>AIC: Akaike information criterion.</p></fn><fn id="table4fn3"><p><sup>c</sup>BIC: Bayesian information criterion.</p></fn><fn id="table4fn4"><p><sup>d</sup>RMSE: root mean square error.</p></fn><fn id="table4fn5"><p><sup>e</sup>MAE: mean absolute error.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s3-5"><title>Composite Fitness-for-Purpose Verdict</title><p>To make the detailed metrics directly usable by decision-makers and applied researchers, the per-data-element results were summarized with the composite traffic-light verdict defined in the Methods section (<xref ref-type="table" rid="table5">Table 5</xref>). For LOS, the verdict was red at all 3 thresholds because the SD changed far beyond the &#x00B1;10% stability window even though the location shift was comparatively small, which is the signature of microaggregation. For ICD vocabulary, the verdict was yellow at k=5, where Jaccard overlap remained above 0.60, and red at k=10 and k=15, where vocabulary loss and frequency redistribution crossed the red thresholds. The verdict therefore communicates a consistent message. The anonymized data of this type support ordinal and comparative use of the diagnosis-to-LOS relationship but are not suitable for analyses that depend on faithful variance structure, accurate absolute LOS estimates, or complete diagnostic vocabulary.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Composite traffic-light fitness-for-purpose verdict per quasi-identifier and k level, with the governing metric values.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Element</td><td align="left" valign="bottom">Values, k</td><td align="left" valign="bottom">Statistic</td><td align="left" valign="bottom">Mean shift (SD change) (days)</td><td align="left" valign="bottom">Cramer <italic>V</italic></td><td align="left" valign="bottom">Verdict<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">LOS<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">5</td><td align="left" valign="top">KS<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup> <italic>D</italic>=0.147</td><td align="left" valign="top">&#x2013;1.81 (&#x2013;6.45)</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup></td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">LOS</td><td align="left" valign="top">10</td><td align="left" valign="top">KS <italic>D</italic>=0.147</td><td align="left" valign="top">&#x2013;1.81 (&#x2013;6.47)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">LOS</td><td align="left" valign="top">15</td><td align="left" valign="top">KS <italic>D</italic>=0.147</td><td align="left" valign="top">&#x2013;1.82 (&#x2013;6.48)</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">ICD<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup></td><td align="left" valign="top">5</td><td align="left" valign="top">Jaccard=0.624</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.062</td><td align="left" valign="top">Yellow</td></tr><tr><td align="left" valign="top">ICD</td><td align="left" valign="top">10</td><td align="left" valign="top">Jaccard=0.493</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.093</td><td align="left" valign="top">Red</td></tr><tr><td align="left" valign="top">ICD</td><td align="left" valign="top">15</td><td align="left" valign="top">Jaccard=0.421</td><td align="left" valign="top">&#x2014;</td><td align="left" valign="top">0.114</td><td align="left" valign="top">Red</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>A red verdict warns that anonymization-induced distortion may alter substantive conclusions; yellow indicates that ordinal or comparative analyses remain feasible with caution.</p></fn><fn id="table5fn2"><p><sup>b</sup>LOS: length of stay.</p></fn><fn id="table5fn3"><p><sup>c</sup>KS: Kolmogorov-Smirnov<italic>.</italic></p></fn><fn id="table5fn4"><p><sup>d</sup>Not applicable.</p></fn><fn id="table5fn5"><p><sup>e</sup>ICD: International Classification of Diseases.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Dose-Response Shape (RQ3)</title><p>Turning to the third RQ, the relationship between k and data-quality change followed a 2-component pattern rather than a single monotonic gradient. For LOS, the KS <italic>D</italic> and its scaled statistic, and all complementary descriptors, were essentially constant across the 3 k levels (<xref ref-type="table" rid="table3">Table 3</xref>) because microaggregation imposes its distributional footprint at the initial step and additional k increases contribute little further perturbation. For the inferential metrics, the same saturation held. The ICC attributable to diagnosis changed only marginally beyond k=5 (0.827, 0.831, and 0.837), BLUP rank concordance was stable (&#x03C1;_S 0.970, 0.966, and 0.964), and Lin CCC changed little (0.966, 0.965, and 0.959). Categorical degradation, by contrast, followed a clear dose-response gradient. Jaccard fell from 0.624 to 0.493 to 0.421, Cramer <italic>V</italic> rose from 0.062 to 0.093 to 0.114, and the number of diagnosis groups no longer estimable scaled with k (234, 345, and 425). The overall pattern is therefore a step-function for the numeric outcome, saturating at k=5, combined with a proportional gradient for categorical erosion, which tracks the suppression of rare codes (<xref ref-type="fig" rid="figure4">Figure 4</xref>). This pattern is a property of the present low-dimensional, 2-quasi-identifier configuration and should not be assumed to hold for higher-dimensional releases, as discussed in the Limitations and Future Work section.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Corroboration of distributional and inferential metrics across k. (A) KS <italic>D</italic> for LOS versus Spearman &#x03C1; of BLUPs, showing minimal variation in distributional shift alongside high concordance. (B) ICD Jaccard versus &#x03C1;. (C) Cramer <italic>V</italic> versus &#x03C1;. The figure shows that increasing k alters distributional and categorical properties more than it alters the reproducibility of relative inferential patterns. BLUP: best linear unbiased prediction; ICD: International Classification of Diseases; KS: Kolmogorov-Smirnov; LOS: length of stay.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e97661_fig04.png"/></fig><p>However, the KS <italic>D</italic> axis in <xref ref-type="fig" rid="figure4">Figure 4A</xref> spans an extremely narrow range because the 3 anonymized LOS distributions are almost identical to one another (KS <italic>D</italic>=0.1467, 0.1470, 0.1470). The points therefore do not order themselves along that axis in a visually monotonic way, and this is expected rather than a sign of instability in the method or the metric. Microaggregation replaces LOS values with cluster means at the first anonymization step, which fixes the shape of the anonymized LOS distribution almost immediately, so that further increases in k change the LOS distribution only negligibly. The near-vertical arrangement of the 3 points along the KS <italic>D</italic> axis is the visual effect of this saturation. The Spearman axis, by contrast, varies smoothly and monotonically because rank concordance is sensitive to the progressive loss of diagnosis groups, which does scale with k. Read together, the 2 axes convey the central finding of the dose-response analysis, namely, that the numeric distortion saturates while the categorical erosion accumulates.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This case study produced 5 central observations. First, the distributional comparison identified a substantial LOS perturbation (KS <italic>D</italic>=0.147, SD compression of approximately 6.5 days, contraction of the upper tail) that the internal loss metric of the anonymization tool did not signal, since that metric reported aggregate geometric-mean losses below 10%. In clinical terms, anonymization shifted the median hospital stay from 4 to 5 days and narrowed the IQR from 5 to 4 days, while the diagnosis-level mean changed only modestly. This combination of a small location change with a large dispersion change would remain invisible to any evaluation relying on first-moment comparison alone, which is the core methodological point of the study.</p><p>Second, the 2 quasi-identifiers showed mechanism-specific degradation patterns. LOS distortion saturated at k=5, whereas ICD vocabulary loss continued to increase with k. These contrasting patterns are compatible with different dominant transformation mechanisms, but because ARX applies both transformations jointly, the present design supports indirect rather than formal attribution and motivates a future factorial decomposition. Third, microaggregation produced an asymmetric moment pattern. The first moment was largely preserved, while the second moment collapsed, and the upper tail was truncated, which is the expected behavior of cluster-centroid substitution applied to a right-skewed distribution. A small number of long-stay patients drives most of the variance and very little of the mean, so collapsing within-cluster variability removes those tail values almost entirely without shifting the typical value. The practical concern is therefore user-side misinterpretation rather than statistical surprise. An analyst comparing only central tendency would conclude that the data are unaffected, while any variance-dependent analysis would silently operate on a homogenized outcome.</p><p>Fourth, the inflation of the ICC after anonymization is most plausibly explained by residual-variance deflation and structural simplification, and it should not be interpreted as evidence of stronger diagnosis-associated information. The interrogation of the variance components showed that the moderate rise in between-diagnosis variance accompanies a sharp fall in the number of diagnosis groups and a near-doubling of median group size, which is consistent with the removal of small, low-frequency groups rather than with any genuine sharpening of the diagnosis-to-LOS signal. Fifth, BLUP rank ordering and aggregate magnitude agreement were both high (&#x03C1;_S 0.964 to 0.970, Lin CCC 0.959 to 0.966, each with narrow CIs). However, approximately 7% of diagnosis categories underwent sign reversals, concentrated almost entirely among diagnoses whose original effect was near 0. The headline message of reproducibility is therefore conditional. Relative ranking is robust, but the effect estimate for any specific low-signal diagnosis can be unreliable.</p></sec><sec id="s4-2"><title>Characteristics of the Suppressed Encounters</title><p>Building on these findings, and because suppression in k-anonymity targets records that fail to reach the k threshold, the masked encounters are not a random sample, and their characteristics were examined directly to assess potential survivor-type bias (analysis of suppressed records.R in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> and gradual_loss_cascade.csv and cohort_characterization.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). Masking was small in volume, affecting under 1% of encounters at the first step and under 3% cumulatively at k=15, but it was nonrandom. It fell mainly on lower-frequency codes, with the largest within-code erosion among musculoskeletal diagnoses (eg, M85 lost 43% of its records, M72 lost 28%, and M65 lost 27%), spinal-cord and related injury (G82 lost 33%), trauma codes (S36 and M84), and one rare infection (A31, with 78% of its records removed at k=15). The masked encounters also had modestly longer LOS than those retained, with mean LOS values of 7.23 (SD 13.08), 7.39 (SD 12.58), and 7.38 (SD 12.21) days for k=5, 10, and 15, respectively, compared with 6.49 (SD 9.65), 6.48 (SD 9.62), and 6.47 (SD 9.60) days among retained encounters and 6.52 (SD 9.74) days overall. However, the median LOS remained 4 days in both groups. Because masking operates at the encounter level rather than the patient level, encounter counts reconcile exactly between the suppressed and retained groups, whereas patient counts do not sum to the cohort total, because a single patient may contribute one encounter that is retained and another that is suppressed and therefore appears in both groups (cohort_characterization.csv in <xref ref-type="supplementary-material" rid="app5">Multimedia Appendix 5</xref>). This outlier-directed, rare code&#x2013;directed pattern explains the survivor-type signal seen elsewhere in the results, and it is reported here as a limitation. Estimates for the specific rare or long-stay conditions most affected by masking warrant particular caution, while the aggregate distributional and rank-level results remain largely unaffected because the masked volume is small.</p></sec><sec id="s4-3"><title>Clinical and Ethical Implications of Vocabulary Erosion</title><p>Beyond the statistical consequences already described, the progressive loss of diagnostic vocabulary carries clinical and ethical implications. At k=15, more than 400 three-character diagnosis categories were no longer estimable, and the analysis above shows that the categories most completely erased are precisely the rare and lower-frequency conditions. Systematically removing rare diagnoses from a research dataset risks rendering those conditions invisible to exactly the secondary analyses that are meant to serve them, such as health-services planning, capacity forecasting, and equity monitoring for uncommon or resource-intensive conditions. If anonymized administrative data are used to allocate resources or to define cohorts, the suppression of rare-disease codes can bias planning toward common conditions and against the long tail of rarer presentations, with the affected patient groups least able to absorb the resulting underrepresentation. This constitutes a fitness-for-purpose concern rather than a privacy concern, and it reinforces the argument that the suppression footprint, and specifically which clinical categories were erased, should be reported alongside any analysis of anonymized clinical data.</p></sec><sec id="s4-4"><title>Implications for Researchers</title><p>Taken together, for medical informaticians and data scientists, these results carry practical consequences. Mixed-model estimates from anonymized data of this type may appear misleadingly precise, with elevated explained-variance values and narrower prediction intervals that mask true patient-level variability. The roughly 10-fold residual-variance compression means that CIs derived from the anonymized data are narrower than warranted, which can inflate apparent statistical significance through a mechanism analogous to range restriction in psychometric research [<xref ref-type="bibr" rid="ref54">54</xref>]. The inflated ICC does not mean that diagnoses explain LOS better after anonymization, only that less unexplained variability remains for them to compete against. Researchers identifying which diagnoses carry the longest or shortest stays would, in this case study, reach concordant rankings from anonymized data, but analyses requiring precise effect magnitudes, such as cost projections, resource-allocation models, or meta-analytic contributions, would be compromised by the magnitude distortion and the low-signal sign reversals documented earlier. The apparent improvement in prediction accuracy should be read in the same light. The lower RMSE on anonymized data reflects a compressed outcome, not a better model, and any prediction metric reported on anonymized data should be accompanied by the corresponding metric on the original data and by the variance ratio of the outcome.</p></sec><sec id="s4-5"><title>Contextualization Within the Literature</title><p>These observations can be situated within the existing literature. Pilgram et al [<xref ref-type="bibr" rid="ref13">13</xref>] evaluated anonymization costs in a chronic kidney disease cohort and reported that use case&#x2013;specific configurations yielded higher reproducibility than generic approaches. The present study extends this by showing that even a generic configuration preserves rank-order inferential structure when evaluated through BLUP concordance, while magnitude fidelity is high in aggregate but fails for individual low-signal diagnoses. Johann et al [<xref ref-type="bibr" rid="ref14">14</xref>] compared anonymized and synthetic heart-failure data and concluded that both degraded utility, which the present granular distributional and quantile-level characterization corroborates and refines. Jakob et al [<xref ref-type="bibr" rid="ref10">10</xref>] evaluated a COVID-19 anonymization pipeline using distributional comparisons but did not test whether clinical associations survived the transformation, a gap addressed here by the mixed-model reproducibility layer. Ohno-Machado et al [<xref ref-type="bibr" rid="ref12">12</xref>] showed that cell suppression preserves descriptive statistics while degrading predictive performance, which the present study sharpens by demonstrating that the diagnosis-level mean shifts modestly while the distributional structure changes substantially. The broader contemporary literature aligns with these findings. Cohen et al [<xref ref-type="bibr" rid="ref15">15</xref>] showed in a clinical data warehouse that reducing reidentification risk required transformations that compromised the reliability of specific epidemiological statistics, and Kaabachi et al [<xref ref-type="bibr" rid="ref16">16</xref>] documented the absence of a single agreed privacy-utility metric set across medical studies. The present contribution is complementary, namely, a transparent, tool-agnostic procedure for documenting which analytical properties a given anonymization configuration preserves and which it does not, and a reporting checklist to make that documentation routine. Kohlmayer et al [<xref ref-type="bibr" rid="ref17">17</xref>] focused on information-theoretic loss functions internal to ARX, and the external evaluation performed here reveals quality dimensions that those internal metrics cannot capture. As with the tool-agnostic anonymity verification of pyCANON [<xref ref-type="bibr" rid="ref58">58</xref>] and related independent utility-assessment approaches [<xref ref-type="bibr" rid="ref59">59</xref>,<xref ref-type="bibr" rid="ref60">60</xref>], fitness-for-purpose evaluation should be independent of the anonymization software used. Finally, the dual-metric strategy of supplementing Spearman correlation with Lin CCC operationalizes the principle of Bland and Altman [<xref ref-type="bibr" rid="ref53">53</xref>] that correlation alone is insufficient for agreement assessment.</p></sec><sec id="s4-6"><title>Limitations and Future Work</title><p>Several limitations should be addressed in future iterations. First, the composite verdict thresholds are based on established statistical conventions rather than calibrated against clinically meaningful differences, and a key next step is to align them with minimal important differences through expert panels and simulation. Second, although a patient-level random effect was added and admission year was used to confirm the absence of temporal drift in suppression, the random-intercept structure does not capture whether anonymization affects specific diagnosis groups differentially; random slopes or use cases with inferential targets less directly exposed to variance depletion would extend the analysis. Third, microaggregation was performed using the geometric mean so that within-cluster aggregation is consistent with the logarithmic scale on which LOS is modeled, which was used to reduce the positive bias that raw-scale arithmetic aggregation would introduce under a concave transformation. We acknowledge that a minor persistent bias may remain because geometric-mean microaggregation aligns with log(LOS), whereas the mixed model was fitted to log(LOS+1), creating a slight scale mismatch particularly at the lowest boundary where the LOS equals 1. A further residual limitation remains that microaggregation of any form compresses within-cluster variance, so mixed models fitted to microaggregated, variance-depleted outcomes may produce near-singular fits and overconfident inference. The compressed residual variance, sharply peaked density plots, very small standard errors, extreme <italic>P</italic> values, and strongly negative AIC values in the anonymized datasets should therefore be interpreted as signs that uncertainty estimates from anonymized data may represent lower bounds rather than fully reliable measures of precision. Fourth, the masking of records is nonrandom and concentrated on rare and longer-stay encounters, so estimates for those specific conditions warrant caution; pattern-mixture or tipping-point sensitivity analyses following Leurent et al [<xref ref-type="bibr" rid="ref61">61</xref>], in the spirit of the missing-not-at-random framework of Little and Rubin [<xref ref-type="bibr" rid="ref62">62</xref>], would test the robustness of the rank-concordance and magnitude results to worst-case assumptions about the masked encounters. Fifth, testing only 3 k values constrains the identification of inflection points, and including lower and higher thresholds, as well as further privacy models such as differential privacy, t-closeness, and l-diversity, would refine the dose-response characterization. Sixth, the evaluation used a single tool, ARX, under one configuration family, so the concept should be understood as a software-agnostic approach illustrated through a single-tool case study; a multitool comparison would substantiate the generalizability claim. Seventh, and most importantly for external validity, the retained dataset deliberately includes only 2 dimensions, primary ICD code and LOS, and excludes variables such as age, sex, and admission and discharge dates. A 2D dataset of this kind has limited standalone utility for epidemiological analyses, patient care-pathway studies, or key-performance-indicator computation even before anonymization, and the saturation of LOS distortion at k=5 is a property of this low-dimensional configuration. Applying k-anonymity to a higher-dimensional release with additional quasi-identifiers would be expected to require more suppression and produce greater utility loss, so the saturation observed here should not be generalized to such settings. Eighth, LOS served simultaneously as a quasi-identifier and as the inferential outcome, a worst-case design that makes part of the observed inferential degradation an upper bound on what analyses unrelated to the transformed quasi-identifiers would experience.</p></sec><sec id="s4-7"><title>Recommendations</title><p>Medical informatics researchers, data integration centers, and data-protection officers should empirically verify the distributional and inferential fitness-for-purpose of any anonymized clinical dataset before secondary use. Before conducting analyses or sharing data, practitioners should apply distributional screening, including quantile-level characterization and variance diagnostics, to quantify the transformation footprint. Inferential reproducibility should then be tested for the specific association under investigation, using both rank-order and magnitude-sensitive agreement measures with CIs. Results from anonymized data should be reported alongside explicit documentation of the k-anonymity level, the complete anonymization configuration (including the transformation model applied to each quasi-identifier, suppression limits, optimization weights, and attacker models), the suppression rate and its distribution across quasi-identifiers and subgroups, the distributional impact metrics, and a statement on whether ICC inflation was observed and how it was interpreted (<xref ref-type="other" rid="box1">Textbox 1</xref> and <xref ref-type="supplementary-material" rid="app6">Checklist 1</xref>).</p><boxed-text id="box1"><title> Suggested reporting-statement template for studies using anonymized clinical data.</title><p>The analysis used data anonymized with [tool name, version] under [privacy model], applying [parameter setting(s)]. Quasi-identifiers were transformed as follows: [variable group 1: method], [variable group 2: method]. The resulting suppression rate was [X]%, distributed [uniformly/non-uniformly] across [subgroups / time]. The impact of anonymization on data utility was quantified with metrics selected to capture distributional preservation, structural or category preservation, and downstream inferential reproducibility, yielding [metric 1] = [X], [metric 2] = [X], and [metric 3] = [X] with 95% confidence intervals. The [agreement metric] changed from [X] in the original dataset to [X] after anonymization. Any apparent increase in agreement should be interpreted cautiously, as it may reflect variance deflation or structural simplification rather than genuinely improved analytical signal. The anonymized data may therefore be less suitable for analyses requiring high-fidelity variance structure, accurate absolute estimates, complete representation of sparse categories, or full preservation of original data granularity. Where the analytical target variable was itself subject to anonymization transformation, this was stated, and its amplifying effect on utility loss was noted.</p></boxed-text><p>To support consistent reporting, we created the ADAQI checklist (Anonymization and Data Analysis Quality and Impact Reporting; <xref ref-type="supplementary-material" rid="app6">Checklist 1</xref>), a structured aid of 29 items organized across 7 reporting domains, and we propose that future publications use it for standardized reporting of anonymization impact. We provide a blank template in <xref ref-type="supplementary-material" rid="app6">Checklist 1</xref> and a worked, filled-in version completed against this study as a concrete example. Completion can be confirmed with a single sentence in the Methods section, for example, anonymization impact was reported using the ADAQI checklist (<xref ref-type="supplementary-material" rid="app6">Checklists 1</xref> and <xref ref-type="supplementary-material" rid="app7">2</xref>).</p></sec><sec id="s4-8"><title>Conclusions</title><p>In summary, this methodological case study demonstrates, using the ARX tool, a procedure that combines distributional fidelity assessment with inferential reproducibility testing for anonymized clinical data. Applied to 719,387 inpatient encounters, k-anonymity at k=5, 10, and 15 produced a measurable distributional change in LOS (KS <italic>D</italic>=0.147, SD compression of about 6.5 days, contraction of the upper tail) and progressive vocabulary erosion (Jaccard declining from 0.624 to 0.421), while preserving the rank ordering of diagnosis-associated LOS effects (Spearman &#x03C1; 0.964-0.970) and the aggregate magnitude agreement (Lin CCC 0.959-0.966). Within this preserved aggregate, a minority of low-signal diagnoses underwent extreme individual distortion, including sign reversals in approximately 7% of categories. None of these effects were signaled by the internal loss metric of the tool, which reported aggregate losses below 10%. The numeric distortion saturated at k=5, while categorical erosion accumulated with k. These findings are specific to a deliberately low-dimensional, 2-quasi-identifier worst-case design, and should not be generalized to higher-dimensional releases. The practical conclusion is that anonymized datasets of this type can support ordinal and comparative analyses of the diagnosis-to-LOS relationship, but can mislead analyses that depend on faithful variance structure, accurate absolute estimates, or complete diagnostic vocabulary. Furthermore, the analytical footprint of anonymization should be measured and reported rather than assumed. The open-source R implementation, the anonymization certificates, and the reporting checklist are available to enable independent replication. Multitool and multidataset validation constitute the next steps of the project.</p></sec></sec></body><back><ack><p>The authors are also grateful to the Data Integration Center Mannheim of the University Hospital Mannheim for data provision and to the Medical Faculty Mannheim of the University of Heidelberg for the institutional support. The authors declare the use of generative AI in the data-analysis and writing process. According to the GAIDeT taxonomy [<xref ref-type="bibr" rid="ref63">63</xref>], the following tasks were delegated to generative AI tools under full human supervision: commenting and structuring R analysis scripts and revising the grammatical correctness of the manuscript text. Manus 1.6 was used to comment and structure R analysis scripts. Grammarly was used to revise the grammatical correctness of the manuscript text. No generative AI tool contributed to study conceptualization, data interpretation, or the formulation of scientific claims. The authors verified all AI-assisted output and retain full responsibility for the entire content of this manuscript.</p></ack><notes><sec><title>Funding</title><p>This publication was partially supported by the German Federal Ministry of Research, Technology and Space (BMFTR) within the Network of University Medicine 3.0 (NUM 3.0; grant 01KX2524).</p></sec><sec><title>Data Availability</title><p>The raw data cannot be publicly shared due to data-protection requirements. Source data have been archived, and access can be made available for individual requests based on approval of the Ethics and Use and Access Committees. Full project documentation, including R source code, function implementation, and anonymization certificates, is publicly available on GitHub [<xref ref-type="bibr" rid="ref31">31</xref>].</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: GKW</p><p>Methodology: GKW</p><p>Software development: GKW</p><p>Validation: TG, FS</p><p>Formal analysis: GKW</p><p>Data curation: GKW</p><p>Supervision: TG, FS</p><p>Project administration: TG, FS</p><p>Funding acquisition: TG, FS</p><p>Writing&#x2014;original draft: GKW</p><p>Writing&#x2014;review and editing: GKW, PPS, MJL, MH, TG, FS</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ADAQI</term><def><p>Anonymization and Data Analysis Quality and Impact Reporting</p></def></def-item><def-item><term id="abb2">AIC</term><def><p>Akaike information criterion</p></def></def-item><def-item><term id="abb3">BLUP</term><def><p>best linear unbiased prediction</p></def></def-item><def-item><term id="abb4">CCC</term><def><p>concordance correlation coefficient</p></def></def-item><def-item><term id="abb5">ICC</term><def><p>intraclass correlation coefficient</p></def></def-item><def-item><term id="abb6">ICD</term><def><p>International Classification of Diseases</p></def></def-item><def-item><term id="abb7">ICD-10-GM</term><def><p>International Classification of Diseases, 10th Revision, German Modification</p></def></def-item><def-item><term id="abb8">ICD-3</term><def><p>Three-character ICD category</p></def></def-item><def-item><term id="abb9">KS</term><def><p>Kolmogorov-Smirnov</p></def></def-item><def-item><term id="abb10">LMM</term><def><p>linear mixed model</p></def></def-item><def-item><term id="abb11">LOS</term><def><p>length of stay</p></def></def-item><def-item><term id="abb12">RMSE</term><def><p>root mean square error</p></def></def-item><def-item><term id="abb13">RQ</term><def><p>research question</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Murdoch</surname><given-names>TB</given-names> </name><name name-style="western"><surname>Detsky</surname><given-names>AS</given-names> </name></person-group><article-title>The inevitable application of big data to health care</article-title><source>JAMA</source><year>2013</year><month>04</month><day>3</day><volume>309</volume><issue>13</issue><fpage>1351</fpage><lpage>1352</lpage><pub-id pub-id-type="doi">10.1001/jama.2013.393</pub-id><pub-id pub-id-type="medline">23549579</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="web"><article-title>Regulation (EU) 2016/679 of the European Parliament and of the Council of 27 April 2016 on the protection of natural persons with regard to the processing of personal data and on the free movement of such data, and repealing directive 95/46/EC (General Data Protection Regulation)</article-title><source>European Union</source><year>2016</year><access-date>2026-09-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng">https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sweeney</surname><given-names>L</given-names> </name></person-group><article-title>k-Anonymity: a model for protecting privacy</article-title><source>Int J Unc Fuzz Knowl Based Syst</source><year>2002</year><month>10</month><volume>10</volume><issue>5</issue><fpage>557</fpage><lpage>570</lpage><pub-id pub-id-type="doi">10.1142/S0218488502001648</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Eicher</surname><given-names>J</given-names> </name><name name-style="western"><surname>Spengler</surname><given-names>H</given-names> </name><name name-style="western"><surname>Bild</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>KA</given-names> </name></person-group><article-title>Flexible data anonymization using ARX&#x2014;current status and challenges ahead</article-title><source>Softw Pract Exp</source><year>2020</year><month>07</month><volume>50</volume><issue>7</issue><fpage>1277</fpage><lpage>1304</lpage><pub-id pub-id-type="doi">10.1002/spe.2812</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Machanavajjhala</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kifer</surname><given-names>D</given-names> </name><name name-style="western"><surname>Gehrke</surname><given-names>J</given-names> </name><name name-style="western"><surname>Venkitasubramaniam</surname><given-names>M</given-names> </name></person-group><article-title>L-diversity: privacy beyond k-anonymity</article-title><source>ACM Trans Knowl Discov Data</source><year>2007</year><volume>1</volume><issue>1</issue><fpage>3</fpage><pub-id pub-id-type="doi">10.1145/1217299.1217302</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>N</given-names> </name><name name-style="western"><surname>Li</surname><given-names>T</given-names> </name><name name-style="western"><surname>Venkatasubramanian</surname><given-names>S</given-names> </name></person-group><article-title>t-Closeness: privacy beyond k-anonymity and l-diversity</article-title><conf-name>2007 IEEE 23rd International Conference on Data Engineering (ICDE 2007)</conf-name><conf-date>Apr 15-20, 2007</conf-date><pub-id pub-id-type="doi">10.1109/ICDE.2007.367856</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dwork</surname><given-names>C</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Bugliesi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Preneel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Sassone</surname><given-names>V</given-names> </name><name name-style="western"><surname>Wegener</surname><given-names>I</given-names> </name></person-group><article-title>Differential privacy</article-title><conf-name>33rd International Colloquium on Automata, Languages and Programming (ICALP 2006)</conf-name><conf-date>Jul 10-14, 2006</conf-date><pub-id pub-id-type="doi">10.1007/11787006_1</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kohlmayer</surname><given-names>F</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Gkoulalas-Divanis</surname><given-names>A</given-names> </name><name name-style="western"><surname>Loukides</surname><given-names>G</given-names> </name></person-group><article-title>Putting statistical disclosure control into practice: the ARX data anonymization tool</article-title><source>Medical Data Privacy Handbook</source><year>2015</year><publisher-name>Springer</publisher-name><fpage>111</fpage><lpage>148</lpage><pub-id pub-id-type="doi">10.1007/978-3-319-23633-9_6</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kohlmayer</surname><given-names>F</given-names> </name><name name-style="western"><surname>Lautenschl&#x00E4;ger</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>KA</given-names> </name></person-group><article-title>ARX&#x2014;a comprehensive tool for anonymizing biomedical data</article-title><source>AMIA Annu Symp Proc</source><year>2014</year><volume>2014</volume><fpage>984</fpage><lpage>993</lpage><pub-id pub-id-type="medline">25954407</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jakob</surname><given-names>CEM</given-names> </name><name name-style="western"><surname>Kohlmayer</surname><given-names>F</given-names> </name><name name-style="western"><surname>Meurers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Vehreschild</surname><given-names>JJ</given-names> </name><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name></person-group><article-title>Design and evaluation of a data anonymization pipeline to promote Open Science on COVID-19</article-title><source>Sci Data</source><year>2020</year><month>12</month><day>10</day><volume>7</volume><issue>1</issue><fpage>435</fpage><pub-id pub-id-type="doi">10.1038/s41597-020-00773-y</pub-id><pub-id pub-id-type="medline">33303746</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kamdje Wabo</surname><given-names>G</given-names> </name><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Gierend</surname><given-names>K</given-names> </name><name name-style="western"><surname>Siegel</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ganslandt</surname><given-names>T</given-names> </name></person-group><article-title>Data quality- and utility-compliant anonymization of common data model-harmonized electronic health record data: protocol for a scoping review</article-title><source>JMIR Res Protoc</source><year>2023</year><month>08</month><day>11</day><volume>12</volume><fpage>e46471</fpage><pub-id pub-id-type="doi">10.2196/46471</pub-id><pub-id pub-id-type="medline">37566443</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ohno-Machado</surname><given-names>L</given-names> </name><name name-style="western"><surname>Vinterbo</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Dreiseitl</surname><given-names>S</given-names> </name></person-group><article-title>Effects of data anonymization by cell suppression on descriptive statistics and predictive modeling performance</article-title><source>J Am Med Inform Assoc</source><year>2002</year><month>11</month><day>1</day><volume>9</volume><issue>90061</issue><fpage>115S</fpage><lpage>119</lpage><pub-id pub-id-type="doi">10.1197/jamia.M1241</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Pilgram</surname><given-names>L</given-names> </name><name name-style="western"><surname>Meurers</surname><given-names>T</given-names> </name><name name-style="western"><surname>Malin</surname><given-names>B</given-names> </name><etal/></person-group><article-title>The costs of anonymization: case study using clinical data</article-title><source>J Med Internet Res</source><year>2024</year><month>04</month><day>24</day><volume>26</volume><fpage>e49445</fpage><pub-id pub-id-type="doi">10.2196/49445</pub-id><pub-id pub-id-type="medline">38657232</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johann</surname><given-names>TI</given-names> </name><name name-style="western"><surname>Otte</surname><given-names>K</given-names> </name><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Dieterich</surname><given-names>C</given-names> </name></person-group><article-title>Anonymize or synthesize? Privacy-preserving methods for heart failure score analytics</article-title><source>Eur Heart J Digit Health</source><year>2025</year><month>01</month><volume>6</volume><issue>1</issue><fpage>147</fpage><lpage>154</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztae083</pub-id><pub-id pub-id-type="medline">39846076</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>A</given-names> </name><name name-style="western"><surname>Jacob</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chatellier</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Quantifying the effects of pseudonymisation on epidemiological research reliability: a tailored evaluation using a clinical data warehouse</article-title><source>BMC Med Inform Decis Mak</source><year>2026</year><month>02</month><day>19</day><volume>26</volume><issue>1</issue><fpage>87</fpage><pub-id pub-id-type="doi">10.1186/s12911-026-03360-0</pub-id><pub-id pub-id-type="medline">41715132</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kaabachi</surname><given-names>B</given-names> </name><name name-style="western"><surname>Despraz</surname><given-names>J</given-names> </name><name name-style="western"><surname>Meurers</surname><given-names>T</given-names> </name><etal/></person-group><article-title>A scoping review of privacy and utility metrics in medical synthetic data</article-title><source>NPJ Digit Med</source><year>2025</year><month>01</month><day>27</day><volume>8</volume><issue>1</issue><fpage>60</fpage><pub-id pub-id-type="doi">10.1038/s41746-024-01359-3</pub-id><pub-id pub-id-type="medline">39870798</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kohlmayer</surname><given-names>F</given-names> </name><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>KA</given-names> </name></person-group><article-title>The cost of quality: Implementing generalization and suppression for anonymizing biomedical data with minimal information loss</article-title><source>J Biomed Inform</source><year>2015</year><month>12</month><volume>58</volume><fpage>37</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.1016/j.jbi.2015.09.007</pub-id><pub-id pub-id-type="medline">26385376</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Busse</surname><given-names>R</given-names> </name><name name-style="western"><surname>Geissler</surname><given-names>A</given-names> </name><name name-style="western"><surname>Aaviksoo</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Diagnosis related groups in Europe: moving towards transparency, efficiency, and quality in hospitals?</article-title><source>BMJ</source><year>2013</year><month>06</month><day>7</day><volume>346</volume><fpage>f3197</fpage><pub-id pub-id-type="doi">10.1136/bmj.f3197</pub-id><pub-id pub-id-type="medline">23747967</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Toh</surname><given-names>HJ</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>ZY</given-names> </name><name name-style="western"><surname>Yap</surname><given-names>P</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>T</given-names> </name></person-group><article-title>Factors associated with prolonged length of stay in older patients</article-title><source>Singapore Med J</source><year>2017</year><month>03</month><volume>58</volume><issue>3</issue><fpage>134</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.11622/smedj.2016158</pub-id><pub-id pub-id-type="medline">27609507</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Harerimana</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>JW</given-names> </name><name name-style="western"><surname>Jang</surname><given-names>B</given-names> </name></person-group><article-title>A deep attention model to forecast the Length Of Stay and the in-hospital mortality right on admission from ICD codes and demographic data</article-title><source>J Biomed Inform</source><year>2021</year><month>06</month><volume>118</volume><fpage>103778</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2021.103778</pub-id><pub-id pub-id-type="medline">33872817</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deschepper</surname><given-names>M</given-names> </name><name name-style="western"><surname>Smedt</surname><given-names>CD</given-names> </name><name name-style="western"><surname>Colpaert</surname><given-names>K</given-names> </name></person-group><article-title>A literature-based approach to predict continuous hospital length of stay in adult acute care patients using admission variables: a single university center experience</article-title><source>Int J Med Inform</source><year>2025</year><month>01</month><volume>193</volume><fpage>105678</fpage><pub-id pub-id-type="doi">10.1016/j.ijmedinf.2024.105678</pub-id><pub-id pub-id-type="medline">39476744</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bottle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gaudoin</surname><given-names>R</given-names> </name><name name-style="western"><surname>Goudie</surname><given-names>R</given-names> </name><name name-style="western"><surname>Jones</surname><given-names>S</given-names> </name><name name-style="western"><surname>Aylin</surname><given-names>P</given-names> </name></person-group><article-title>Can valid and practical risk-prediction or casemix adjustment models, including adjustment for comorbidity, be generated from English hospital administrative data (Hospital Episode Statistics)? A national observational study</article-title><source>Health Serv Deliv Res</source><year>2014</year><volume>2</volume><issue>40</issue><fpage>1</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.3310/hsdr02400</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Johnson</surname><given-names>T</given-names> </name><name name-style="western"><surname>McNutt</surname><given-names>R</given-names> </name><name name-style="western"><surname>Odwazny</surname><given-names>R</given-names> </name><name name-style="western"><surname>Patel</surname><given-names>D</given-names> </name><name name-style="western"><surname>Baker</surname><given-names>S</given-names> </name></person-group><article-title>Discrepancy between admission and discharge diagnoses as a predictor of hospital length of stay</article-title><source>J Hosp Med</source><year>2009</year><month>04</month><volume>4</volume><issue>4</issue><fpage>234</fpage><lpage>239</lpage><pub-id pub-id-type="doi">10.1002/jhm.453</pub-id><pub-id pub-id-type="medline">19388065</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="web"><person-group person-group-type="author"><collab>Healthcare Cost and Utilization Project (HCUP)</collab></person-group><article-title>Introduction to the HCUP National Inpatient Sample (NIS)</article-title><source>Agency for Healthcare Research and Quality</source><year>2018</year><access-date>2026-07-05</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.hcup-us.ahrq.gov/db/nation/nis/NIS_Introduction_2018.jsp">https://www.hcup-us.ahrq.gov/db/nation/nis/NIS_Introduction_2018.jsp</ext-link></comment></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Miazgowski</surname><given-names>B</given-names> </name><name name-style="western"><surname>Pakulski</surname><given-names>C</given-names> </name><name name-style="western"><surname>Miazgowski</surname><given-names>T</given-names> </name></person-group><article-title>Length of stay in emergency department by ICD-10 specific and non-specific diagnoses: a single-centre retrospective study</article-title><source>J Clin Med</source><year>2023</year><month>07</month><day>14</day><volume>12</volume><issue>14</issue><fpage>4679</fpage><pub-id pub-id-type="doi">10.3390/jcm12144679</pub-id><pub-id pub-id-type="medline">37510793</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferr&#x00E3;o</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Oliveira</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Gartner</surname><given-names>D</given-names> </name><name name-style="western"><surname>Janela</surname><given-names>F</given-names> </name><name name-style="western"><surname>Martins</surname><given-names>HMG</given-names> </name></person-group><article-title>Leveraging electronic health record data to inform hospital resource management: a systematic data mining approach</article-title><source>Health Care Manag Sci</source><year>2021</year><month>12</month><volume>24</volume><issue>4</issue><fpage>716</fpage><lpage>741</lpage><pub-id pub-id-type="doi">10.1007/s10729-021-09554-4</pub-id><pub-id pub-id-type="medline">34031792</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="web"><article-title>Definitionshandbuch 2024</article-title><source>InEK GmbH</source><year>2024</year><access-date>2026-07-05</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.g-drg.de/ag-drg-system-2024/definitionshandbuch/definitionshandbuch-2024">https://www.g-drg.de/ag-drg-system-2024/definitionshandbuch/definitionshandbuch-2024</ext-link></comment></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quan</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sundararajan</surname><given-names>V</given-names> </name><name name-style="western"><surname>Halfon</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Coding algorithms for defining comorbidities in ICD-9-CM and ICD-10 administrative data</article-title><source>Med Care</source><year>2005</year><month>11</month><volume>43</volume><issue>11</issue><fpage>1130</fpage><lpage>1139</lpage><pub-id pub-id-type="doi">10.1097/01.mlr.0000182534.19832.83</pub-id><pub-id pub-id-type="medline">16224307</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bottle</surname><given-names>A</given-names> </name><name name-style="western"><surname>Aylin</surname><given-names>P</given-names> </name></person-group><article-title>Comorbidity scores for administrative data benefited from adaptation to local coding and diagnostic practices</article-title><source>J Clin Epidemiol</source><year>2011</year><month>12</month><volume>64</volume><issue>12</issue><fpage>1426</fpage><lpage>1433</lpage><pub-id pub-id-type="doi">10.1016/j.jclinepi.2011.04.004</pub-id><pub-id pub-id-type="medline">21764557</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bild</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kuhn</surname><given-names>KA</given-names> </name><name name-style="western"><surname>Prasser</surname><given-names>F</given-names> </name></person-group><article-title>SafePub: a truthful data anonymization algorithm with strong privacy guarantees</article-title><source>Proc Priv Enhancing Technol</source><year>2018</year><month>01</month><day>1</day><volume>2018</volume><issue>1</issue><fpage>67</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1515/popets-2018-0004</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="web"><person-group person-group-type="author"><name name-style="western"><surname>Kamdje Wabo</surname><given-names>G</given-names> </name></person-group><article-title>Quantifying the impact of k-anonymization on clinical data quality</article-title><source>GitHub</source><access-date>2026-06-17</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/gaetankamdje-wabo/Quantifying-the-Impact-of-k-Anonymization-on-Clinical-Data-Quality">https://github.com/gaetankamdje-wabo/Quantifying-the-Impact-of-k-Anonymization-on-Clinical-Data-Quality</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kolmogorov</surname><given-names>A</given-names> </name></person-group><article-title>On the empirical determination of a distribution law</article-title><source>Giorn Ist Ital Attuari</source><year>1933</year><access-date>2026-09-03</access-date><volume>4</volume><fpage>83</fpage><lpage>91</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://cir.nii.ac.jp/crid/1571135650766370304">https://cir.nii.ac.jp/crid/1571135650766370304</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Berger</surname><given-names>VW</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>Y</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Balakrishnan</surname><given-names>N</given-names> </name><name name-style="western"><surname>Colton</surname><given-names>T</given-names> </name><name name-style="western"><surname>Everitt</surname><given-names>B</given-names> </name></person-group><article-title>Kolmogorov-Smirnov test: overview</article-title><source>Wiley StatsRef: Statistics Reference Online</source><year>2014</year><publisher-name>Wiley</publisher-name><pub-id pub-id-type="doi">10.1002/9781118445112</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Conover</surname><given-names>WJ</given-names> </name></person-group><source>Practical Nonparametric Statistics</source><year>1999</year><edition>3</edition><publisher-name>Wiley</publisher-name><pub-id pub-id-type="other">9780471160687</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Massey</surname><given-names>FJ</given-names> </name></person-group><article-title>The Kolmogorov-Smirnov test for goodness of fit</article-title><source>J Am Stat Assoc</source><year>1951</year><month>03</month><volume>46</volume><issue>253</issue><fpage>68</fpage><lpage>78</lpage><pub-id pub-id-type="doi">10.1080/01621459.1951.10500769</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fay</surname><given-names>MP</given-names> </name><name name-style="western"><surname>Proschan</surname><given-names>MA</given-names> </name></person-group><article-title>Wilcoxon-Mann-Whitney or t-test? On assumptions for hypothesis tests and multiple interpretations of decision rules</article-title><source>Stat Surv</source><year>2010</year><volume>4</volume><fpage>1</fpage><lpage>39</lpage><pub-id pub-id-type="doi">10.1214/09-SS051</pub-id><pub-id pub-id-type="medline">20414472</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jaccard</surname><given-names>P</given-names> </name></person-group><article-title>Comparative study of floral distribution in a portion of the Alps and the Jura</article-title><source>Bull Soc Vaudoise Sci Nat</source><year>1901</year><volume>37</volume><fpage>547</fpage><lpage>579</lpage><pub-id pub-id-type="doi">10.5169/seals-266450</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Tanimoto</surname><given-names>TT</given-names> </name></person-group><source>An Elementary Mathematical Theory of Classification and Prediction</source><year>1958</year><publisher-name>International Business Machines Corporation</publisher-name></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Cramer</surname><given-names>H</given-names> </name></person-group><source>Mathematical Methods of Statistics</source><year>1946</year><publisher-name>Princeton University Press</publisher-name><pub-id pub-id-type="other">9780691080048</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>J</given-names> </name></person-group><source>Statistical Power Analysis for the Behavioral Sciences</source><year>1988</year><edition>2</edition><publisher-name>Erlbaum</publisher-name><pub-id pub-id-type="doi">10.4324/9780203771587</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Agresti</surname><given-names>A</given-names> </name></person-group><source>Categorical Data Analysis</source><year>2013</year><edition>3</edition><publisher-name>Wiley</publisher-name><pub-id pub-id-type="other">9780470463635</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Westgard</surname><given-names>JO</given-names> </name></person-group><article-title>Internal quality control: planning and implementation strategies</article-title><source>Ann Clin Biochem</source><year>2003</year><month>11</month><volume>40</volume><issue>Pt 6</issue><fpage>593</fpage><lpage>611</lpage><pub-id pub-id-type="doi">10.1258/000456303770367199</pub-id><pub-id pub-id-type="medline">14629798</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Haizoune</surname><given-names>M</given-names> </name><name name-style="western"><surname>Leventhal</surname><given-names>BL</given-names> </name><name name-style="western"><surname>Pant</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Balancing privacy and utility in child and adolescent mental health services research: retrospective cohort study on synthetic data generation</article-title><source>JMIR Med Inform</source><year>2026</year><month>02</month><day>26</day><volume>14</volume><fpage>e71819</fpage><pub-id pub-id-type="doi">10.2196/71819</pub-id><pub-id pub-id-type="medline">41747250</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Pinheiro</surname><given-names>JC</given-names> </name><name name-style="western"><surname>Bates</surname><given-names>DM</given-names> </name></person-group><source>Mixed-Effects Models in S and S-PLUS</source><year>2000</year><publisher-name>Springer</publisher-name><pub-id pub-id-type="doi">10.1007/b98882</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Robinson</surname><given-names>GK</given-names> </name></person-group><article-title>That BLUP is a good thing: the estimation of random effects</article-title><source>Statist Sci</source><year>1991</year><volume>6</volume><issue>1</issue><fpage>15</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.1214/ss/1177011926</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bates</surname><given-names>D</given-names> </name><name name-style="western"><surname>Maechler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bolker</surname><given-names>B</given-names> </name><name name-style="western"><surname>Walker</surname><given-names>S</given-names> </name></person-group><article-title>Fitting linear mixed-effects models using lme4</article-title><source>J Stat Softw</source><year>2015</year><fpage>1</fpage><lpage>48</lpage><pub-id pub-id-type="doi">10.18637/jss.v067.i01</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patterson</surname><given-names>HD</given-names> </name><name name-style="western"><surname>Thompson</surname><given-names>R</given-names> </name></person-group><article-title>Recovery of inter-block information when block sizes are unequal</article-title><source>Biometrika</source><year>1971</year><volume>58</volume><issue>3</issue><fpage>545</fpage><lpage>554</lpage><pub-id pub-id-type="doi">10.1093/biomet/58.3.545</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shrout</surname><given-names>PE</given-names> </name><name name-style="western"><surname>Fleiss</surname><given-names>JL</given-names> </name></person-group><article-title>Intraclass correlations: uses in assessing rater reliability</article-title><source>Psychol Bull</source><year>1979</year><volume>86</volume><issue>2</issue><fpage>420</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1037/0033-2909.86.2.420</pub-id><pub-id pub-id-type="medline">18839484</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nakagawa</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schielzeth</surname><given-names>H</given-names> </name></person-group><article-title>A general and simple method for obtaining R-squared from generalized linear mixed-effects models</article-title><source>Methods Ecol Evol</source><year>2013</year><volume>4</volume><issue>2</issue><fpage>133</fpage><lpage>142</lpage><pub-id pub-id-type="doi">10.1111/j.2041-210x.2012.00261.x</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Spearman</surname><given-names>C</given-names> </name></person-group><article-title>The proof and measurement of association between two things</article-title><source>Am J Psychol</source><year>1904</year><month>01</month><volume>15</volume><issue>1</issue><fpage>72</fpage><pub-id pub-id-type="doi">10.2307/1412159</pub-id></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>LI</given-names> </name></person-group><article-title>A concordance correlation coefficient to evaluate reproducibility</article-title><source>Biometrics</source><year>1989</year><month>03</month><volume>45</volume><issue>1</issue><fpage>255</fpage><lpage>268</lpage><pub-id pub-id-type="doi">10.2307/2532051</pub-id><pub-id pub-id-type="medline">2720055</pub-id></nlm-citation></ref><ref id="ref52"><label>52</label><nlm-citation citation-type="report"><person-group person-group-type="author"><name name-style="western"><surname>McBride</surname><given-names>GB</given-names> </name></person-group><article-title>A proposal for strength-of-agreement criteria for Lin&#x2019;s concordance correlation coefficient</article-title><year>2005</year><access-date>2026-07-05</access-date><publisher-name>National Institute of Water and Atmospheric Research</publisher-name><comment>NIWA Client Report HAM2005-062</comment><comment><ext-link ext-link-type="uri" xlink:href="https://www.medcalc.org/download/pdf/McBride2005.pdf">https://www.medcalc.org/download/pdf/McBride2005.pdf</ext-link></comment></nlm-citation></ref><ref id="ref53"><label>53</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bland</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Altman</surname><given-names>DG</given-names> </name></person-group><article-title>Statistical methods for assessing agreement between two methods of clinical measurement</article-title><source>Lancet</source><year>1986</year><month>02</month><day>8</day><volume>1</volume><issue>8476</issue><fpage>307</fpage><lpage>310</lpage><pub-id pub-id-type="medline">2868172</pub-id></nlm-citation></ref><ref id="ref54"><label>54</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Gelman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hill</surname><given-names>J</given-names> </name></person-group><source>Data Analysis Using Regression and Multilevel/Hierarchical Models</source><year>2007</year><publisher-name>Cambridge University Press</publisher-name><pub-id pub-id-type="doi">10.1017/CBO9780511790942</pub-id></nlm-citation></ref><ref id="ref55"><label>55</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Snijders</surname><given-names>TAB</given-names> </name><name name-style="western"><surname>Bosker</surname><given-names>RJ</given-names> </name></person-group><source>Multilevel Analysis: An Introduction to Basic and Advanced Multilevel Modeling</source><year>2012</year><edition>2</edition><publisher-name>SAGE</publisher-name><pub-id pub-id-type="other">9781849202015</pub-id></nlm-citation></ref><ref id="ref56"><label>56</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Austin</surname><given-names>PC</given-names> </name><name name-style="western"><surname>Merlo</surname><given-names>J</given-names> </name></person-group><article-title>Intermediate and advanced topics in multilevel logistic regression analysis</article-title><source>Stat Med</source><year>2017</year><month>09</month><day>10</day><volume>36</volume><issue>20</issue><fpage>3257</fpage><lpage>3277</lpage><pub-id pub-id-type="doi">10.1002/sim.7336</pub-id><pub-id pub-id-type="medline">28543517</pub-id></nlm-citation></ref><ref id="ref57"><label>57</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Goldstein</surname><given-names>H</given-names> </name></person-group><source>Multilevel Statistical Models</source><year>2011</year><edition>4</edition><publisher-name>Wiley</publisher-name><pub-id pub-id-type="doi">10.1002/9780470973394</pub-id></nlm-citation></ref><ref id="ref58"><label>58</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>S&#x00E1;inz-Pardo D&#x00ED;az</surname><given-names>J</given-names> </name><name name-style="western"><surname>L&#x00F3;pez Garc&#x00ED;a</surname><given-names>&#x00C1;</given-names> </name></person-group><article-title>A Python library to check the level of anonymity of a dataset</article-title><source>Sci Data</source><year>2022</year><month>12</month><day>26</day><volume>9</volume><issue>1</issue><fpage>785</fpage><pub-id pub-id-type="doi">10.1038/s41597-022-01894-2</pub-id><pub-id pub-id-type="medline">36572676</pub-id></nlm-citation></ref><ref id="ref59"><label>59</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hammer</surname><given-names>FGH</given-names> </name><name name-style="western"><surname>Buglowski</surname><given-names>M</given-names> </name><name name-style="western"><surname>Stollenwerk</surname><given-names>A</given-names> </name></person-group><article-title>Semi-local time sensitive anonymization of clinical data</article-title><source>Sci Data</source><year>2024</year><month>12</month><day>20</day><volume>11</volume><issue>1</issue><fpage>1412</fpage><pub-id pub-id-type="doi">10.1038/s41597-024-04192-1</pub-id><pub-id pub-id-type="medline">39706828</pub-id></nlm-citation></ref><ref id="ref60"><label>60</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ferr&#x00E3;o</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Prata</surname><given-names>P</given-names> </name><name name-style="western"><surname>Fazendeiro</surname><given-names>P</given-names> </name></person-group><article-title>Utility-driven assessment of anonymized data via clustering</article-title><source>Sci Data</source><year>2022</year><month>07</month><day>30</day><volume>9</volume><issue>1</issue><fpage>456</fpage><pub-id pub-id-type="doi">10.1038/s41597-022-01561-6</pub-id><pub-id pub-id-type="medline">35907927</pub-id></nlm-citation></ref><ref id="ref61"><label>61</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Leurent</surname><given-names>B</given-names> </name><name name-style="western"><surname>Gomes</surname><given-names>M</given-names> </name><name name-style="western"><surname>Faria</surname><given-names>R</given-names> </name><name name-style="western"><surname>Morris</surname><given-names>S</given-names> </name><name name-style="western"><surname>Grieve</surname><given-names>R</given-names> </name><name name-style="western"><surname>Carpenter</surname><given-names>JR</given-names> </name></person-group><article-title>Sensitivity analysis for not-at-random missing data in trial-based cost-effectiveness analysis: a tutorial</article-title><source>Pharmacoeconomics</source><year>2018</year><month>08</month><volume>36</volume><issue>8</issue><fpage>889</fpage><lpage>901</lpage><pub-id pub-id-type="doi">10.1007/s40273-018-0650-5</pub-id><pub-id pub-id-type="medline">29679317</pub-id></nlm-citation></ref><ref id="ref62"><label>62</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Little</surname><given-names>RJA</given-names> </name><name name-style="western"><surname>Rubin</surname><given-names>DB</given-names> </name></person-group><source>Statistical Analysis with Missing Data</source><year>2019</year><edition>3</edition><publisher-name>Wiley</publisher-name><pub-id pub-id-type="doi">10.1002/9781119482260</pub-id></nlm-citation></ref><ref id="ref63"><label>63</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Suchikova</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tsybuliak</surname><given-names>N</given-names> </name><name name-style="western"><surname>Teixeira da Silva</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Nazarovets</surname><given-names>S</given-names> </name></person-group><article-title>GAIDeT (Generative AI Delegation Taxonomy): a taxonomy for humans to delegate tasks to generative artificial intelligence in scientific research and publishing</article-title><source>Account Res</source><year>2026</year><month>04</month><volume>33</volume><issue>3</issue><fpage>1</fpage><lpage>27</lpage><pub-id pub-id-type="doi">10.1080/08989621.2025.2544331</pub-id><pub-id pub-id-type="medline">40781729</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>R Project.</p><media xlink:href="medinform_v14i1e97661_app1.zip" xlink:title="ZIP File, 58 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>ARX certificate for k-anonymized data with k=5.</p><media xlink:href="medinform_v14i1e97661_app2.pdf" xlink:title="PDF File, 31 KB"/></supplementary-material><supplementary-material id="app3"><label>Multimedia Appendix 3</label><p>ARX certificate for k-anonymized data with k=10.</p><media xlink:href="medinform_v14i1e97661_app3.pdf" xlink:title="PDF File, 31 KB"/></supplementary-material><supplementary-material id="app4"><label>Multimedia Appendix 4</label><p>ARX certificate for k-anonymized data with k=15.</p><media xlink:href="medinform_v14i1e97661_app4.pdf" xlink:title="PDF File, 31 KB"/></supplementary-material><supplementary-material id="app5"><label>Multimedia Appendix 5</label><p>Detailed plots and tables.</p><media xlink:href="medinform_v14i1e97661_app5.zip" xlink:title="ZIP File, 1129 KB"/></supplementary-material><supplementary-material id="app6"><label>Checklist 1</label><p>ADAQI checklist template.</p><media xlink:href="medinform_v14i1e97661_app6.pdf" xlink:title="PDF File, 181 KB"/></supplementary-material><supplementary-material id="app7"><label>Checklist 2</label><p>ADAQI checklist filled out.</p><media xlink:href="medinform_v14i1e97661_app7.pdf" xlink:title="PDF File, 164 KB"/></supplementary-material></app-group></back></article>