<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e93680</article-id><article-id pub-id-type="doi">10.2196/93680</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Network Analysis&#x2013;Driven Machine Learning Model for Identifying High-Cost Stroke Inpatients Using Hospital Discharge Data: Retrospective Study</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Shen</surname><given-names>Haohui</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author" equal-contrib="yes"><name name-style="western"><surname>Yang</surname><given-names>Yilong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">*</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Mengge</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xiang</surname><given-names>Jingyi</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Runan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Yao</surname><given-names>Pin</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Health Policy and Management, Hangzhou Normal University</institution><addr-line>Hangzhou</addr-line><addr-line>Zhejiang</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Pediatrics, Shengjing Hospital of China Medical University</institution><addr-line>Liaoning</addr-line><addr-line>Shenyang</addr-line><country>China</country></aff><aff id="aff3"><institution>Department of Health Management, Shenyang Women's and Children's Hospital</institution><addr-line>No.87 Danan Road</addr-line><addr-line>Shenyang</addr-line><addr-line>Liaoning</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Staffini</surname><given-names>Alessio</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Phuyal</surname><given-names>Sudip</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Pin Yao, PhD, Department of Health Management, Shenyang Women's and Children's Hospital, No.87 Danan Road, Shenyang, Liaoning, 110011, China, +86-18940056788, +86-024-22853728; <email>yaopincmu@163.com</email></corresp><fn fn-type="equal" id="equal-contrib1"><label>*</label><p>these authors contributed equally</p></fn></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>10</day><month>8</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e93680</elocation-id><history><date date-type="received"><day>19</day><month>02</month><year>2026</year></date><date date-type="rev-recd"><day>07</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>08</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Haohui Shen, Yilong Yang, Mengge Zhang, Jingyi Xiang, Runan Wang, Pin Yao. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 10.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e93680"/><abstract><sec><title>Background</title><p>The medical burden caused by stroke is increasingly severe, and a small minority of high-cost patients consume the majority of medical expenditures. Therefore, revealing the formation mechanisms of this population and exploring a scientific cost-risk stratification system are crucial for improving the quality of care and achieving the optimal allocation of medical resources.</p></sec><sec><title>Objective</title><p>This study aimed to construct a comorbidity network for patients with stroke using standardized front-page medical record data, extract network features that reflect complex disease interactions, and develop identification models in combination with machine learning algorithms. The study focused on building a core model integrating variables from the near-discharge stage for stratifying the risk of high hospitalization costs in patients at the near-discharge stage. In addition, an early prediction model was developed using only data available at admission.</p></sec><sec sec-type="methods"><title>Methods</title><p>We conducted a retrospective study, collecting the hospital discharge data of inpatients with stroke from a tertiary hospital in Northeast China between 2021 and 2023. The data from 2021 to 2022 were used to construct a network and extract features to capture the potential relationship between diseases and high costs. Using the 2023 data partitioned into training and testing sets, we developed 5 models to identify inpatients with stroke who incurred high hospitalization costs and compared their performance when input with different features. In addition, the Shapley Additive Explanations interpretability method was adopted to explain the global and local contributions of the model features.</p></sec><sec sec-type="results"><title>Results</title><p>The inclusion of network features significantly improved the model&#x2019;s performance, among which Extreme Gradient Boosting performed the best. The global feature importance showed that network features occupied a major proportion. The results of the Shapley Additive Explanations interaction analysis indicated potential phased changes in patient resource consumption. However, the overall performance of the early identification model constructed solely from admission data was subject to clear limitations.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study developed an integrated framework combining comorbidity network analysis with machine learning, which significantly improved the accuracy of identifying inpatients with stroke at high risk of incurring excessive hospitalization costs. The core model demonstrated good performance in risk stratification during the near-discharge stage, showing potential for application in the formulation of risk management strategies and the optimization of health care resource allocation. It also laid the foundation for the subsequent development of more accurate early identification models.</p></sec></abstract><kwd-group><kwd>stroke</kwd><kwd>high-cost patients</kwd><kwd>machine learning</kwd><kwd>comorbidity network</kwd><kwd>network analysis</kwd><kwd>hospital discharge data</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>According to the 2021 Global Burden of Disease study, stroke remains the second leading cause of death worldwide [<xref ref-type="bibr" rid="ref1">1</xref>]. It is estimated that the number of stroke-related deaths will increase by 50% between 2020 and 2050, with this health burden predominantly borne by low- and middle-income countries [<xref ref-type="bibr" rid="ref2">2</xref>]. In China, the economic burden of stroke is particularly heavy; in 2018, its direct treatment costs reached $58.6 billion, of which hospitalization expenses accounted for 79.03% [<xref ref-type="bibr" rid="ref3">3</xref>]. Notably, the consumption of medical resources typically exhibits a significantly skewed distribution, with disproportionate medical expenditures often concentrated within a small group of high-cost patients [<xref ref-type="bibr" rid="ref4">4</xref>]. This phenomenon of cost concentration is prevalent worldwide [<xref ref-type="bibr" rid="ref5">5</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Although interventions, such as interdisciplinary transitional care and complex care management, aim to control costs by optimizing services [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref11">11</xref>] and have been proven to have certain short-term effects [<xref ref-type="bibr" rid="ref12">12</xref>,<xref ref-type="bibr" rid="ref13">13</xref>], many high-cost patients still face issues of overtreatment or inefficient care [<xref ref-type="bibr" rid="ref14">14</xref>]. In this context, implementing rational risk stratification according to patients&#x2019; overall disease conditions could potentially address existing shortcomings, ultimately leading to improved cost containment and the optimal allocation of health care resources.</p><p>Owing to its powerful data processing capacity, machine learning has been extensively used in research concerning health care expenditure prediction. For example, Ma et al [<xref ref-type="bibr" rid="ref15">15</xref>] predicted the average daily costs of patients with psychiatric disorders, Hu et al [<xref ref-type="bibr" rid="ref16">16</xref>] identified the determinants of high costs among patients with breast cancer, and Osawa et al [<xref ref-type="bibr" rid="ref17">17</xref>] developed a predictive model for high-need, high-cost patients. Although previous studies have confirmed the utility of machine learning in predicting medical costs and identifying high-cost patients, it remains necessary to extract latent features closely related to medical expenditures from limited datasets to further enhance model performance. Stroke is typically characterized by a high incidence of multiple comorbidities [<xref ref-type="bibr" rid="ref18">18</xref>], making the effective use of patients&#x2019; rich comprehensive diagnostic information crucial. Complex comorbidities not only increase the difficulty of treatment and the risk of mortality but are also closely associated with significant escalations in hospitalization costs [<xref ref-type="bibr" rid="ref19">19</xref>,<xref ref-type="bibr" rid="ref20">20</xref>]. Although existing predictive models generally incorporate comorbidities as key features [<xref ref-type="bibr" rid="ref21">21</xref>-<xref ref-type="bibr" rid="ref23">23</xref>] regarding feature processing, most studies rely on rule-based scoring systems (eg, the Charlson comorbidity index [CCI] and the Elixhauser comorbidity index [ECI]) [<xref ref-type="bibr" rid="ref24">24</xref>,<xref ref-type="bibr" rid="ref25">25</xref>] or simple disease counts. These traditional methods focus solely on the linear accumulation of comorbidities, ignoring the intricate interactions among diseases, which leaves latent information in the data untapped and thereby limits the models&#x2019; ability to identify high-cost patients.</p><p>To overcome the aforementioned methodological limitations, network analysis provides a systematic methodological framework. This network-based analytical paradigm goes beyond traditional simple disease counting; by integrating multiple quantitative indicators such as correlation coefficients, odds ratios (ORs), and the Salton cosine index [<xref ref-type="bibr" rid="ref26">26</xref>], it can effectively evaluate the strength of associations between diseases and subsequently construct comorbidity networks [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>]. This topological perspective not only helps identify frequently co-occurring disease clusters but also further reveals the potential interactions between diseases. With the widespread application and standardization of the <italic>International Classification of Diseases, Tenth Revision</italic> (<italic>ICD-10</italic>) codes, it has become possible to construct phenotypic comorbidity networks (PCNs) using hospital discharge data. For example, Xu et al [<xref ref-type="bibr" rid="ref29">29</xref>] improved the prediction accuracy of self-harm behavior by incorporating comorbidity network features, while Hu et al [<xref ref-type="bibr" rid="ref30">30</xref>] effectively predicted patients&#x2019; length of stay (LOS) using a multiplex network and a patient similarity network. Notably, Yang et al [<xref ref-type="bibr" rid="ref31">31</xref>] constructed a dual network using the diagnostic records of patients with ischemic heart disease, revealing that the extracted network features significantly outperformed traditional comorbidity indices, thereby effectively enhancing the model&#x2019;s ability to identify high-cost patients.</p><p>To the best of our knowledge, network analysis has not yet been applied to the identification and risk characterization of high-cost inpatients with stroke. Given that stroke, as the second leading cause of death globally, imposes a profound resource burden on health care systems, addressing this research gap holds significant practical relevance. Furthermore, previous studies have largely focused on improving overall model performance while often neglecting the interpretability of internal decision-making mechanisms, thereby failing to fully elucidate the specific contributions of individual features to medical costs and their association patterns. Therefore, this study aimed to construct a comorbidity network based on data from the medical record front pages of discharged patients, extract topological features that characterize the complex interactions among diseases, and subsequently build a near-discharge classifier to verify the applicability of network analysis in this research field. Simultaneously, by introducing the Shapley Additive Explanations (SHAP) analysis method, this study is dedicated to developing a robust and clinically interpretable evaluation framework, aiming to provide valuable references for the cost-risk stratification of patients with stroke, thereby offering a scientific basis for hospital administrators to formulate more refined and targeted cost control strategies.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Data Sources and Preprocessing</title><p>This retrospective study used discharge data from a tertiary hospital in Northeast China. The data collection period was from January 1, 2021, to December 31, 2023. The dataset contains comprehensive information on inpatients, including demographic characteristics (such as age and gender), diagnostic information at admission, including primary and secondary diagnoses, admission and discharge status, and total hospitalization costs. On the basis of the <italic>ICD-10</italic> coding system, this study screened for patients with stroke and ultimately included inpatient cases with a primary diagnosis code of I60, I61, or I63. To minimize information bias and ensure data quality, this study established a rigorous data screening and preprocessing workflow. During the data quality control phase, extreme cases with a LOS of less than 24 hours or more than 90 days were first excluded. Subsequently, a systematic evaluation of the missingness of key features was conducted. To balance data integrity and sample representativeness, as well as to prevent the introduction of potential systematic bias from overreliance on data imputation, variables demonstrating a missingness rate greater than 20% were eliminated [<xref ref-type="bibr" rid="ref32">32</xref>]. Among the initially extracted raw features, a total of 2 variables were excluded for exceeding this threshold; the remaining variables exhibited good data completeness, with missing proportions ranging from 0.13% to 1.95%. Statistical evaluation confirmed that the missing data in this study satisfied the assumption of missing at random, and no statistically significant differences in missingness rates were detected across the various cost groups. To impute the remaining missing values, this research used the K-Nearest Neighbors (KNN) imputation technique, a method that estimates missing data by leveraging local structural similarities across samples, thereby effectively retaining the original data&#x2019;s distributional properties and the underlying associations among variables. To avoid data leakage from distance-based imputation, we fitted the KNN imputer using only the 2021 to 2022 network construction cohort and the 2023 training set and then applied the trained fixed model to the 2023 test set for missing value imputation, which ensured complete isolation of data between the training and test sets.</p><p>As readmission within 30 days of discharge typically exhibits a strong clinical correlation with the initial hospitalization, it is considered to reflect continuous disease progression following the primary admission. Therefore, for patients with multiple hospitalization records, this study referred to the processing methods of previous health services research and merged the data using a 30-day time window rule based on unique inpatient numbers and admission and discharge dates [<xref ref-type="bibr" rid="ref33">33</xref>,<xref ref-type="bibr" rid="ref34">34</xref>]. The specific record integration rule was as follows: if the time interval between a patient&#x2019;s previous discharge and subsequent admission was &#x2264;30 days, they were merged into a single hospitalization event; if the time interval was greater than 30 days, they were treated as independent inpatient cases. For cases with multiple consecutive hospitalizations where the intervals between all adjacent admissions met the &#x2264;30-day condition, this study adopted a continuous sequential merging strategy, unifying them into one complete medical event. When merging specific variables, the following principles were adhered to: for features such as gender, age, insurance type, admission year, and primary diagnosis, data from the first record within the merged period were extracted; the admission route was determined using a priority rule (ie, prioritizing emergency admissions, followed by outpatient, and finally other routes); regarding the planned readmission within 31 days indicator, if any record within the merged group was marked as &#x201C;yes,&#x201D; the entire merged event was classified as &#x201C;yes&#x201D;; and the discharge disposition was based on the status at the time of the final discharge. For secondary diagnostic codes, the system aggregated all records within the group and performed deduplication to construct a complete, comprehensive diagnostic set for the patient. Furthermore, the total hospitalization costs and LOS were calculated by summing all records within the merged group. In this merging process, a total of 396 patients accounted for 1096 repeat hospitalization records; through the aforementioned rules, 518 records were retained individually, and the remaining records were merged into 132, ultimately forming 650 hospitalization event records.</p><p>Following the aforementioned screening and processing, this study ultimately included 10,556 patients with stroke, of which 6618 were from 2021 to 2022 and 3938 were from 2023. To prevent data leakage, this study adopted a time span&#x2013;based data partitioning strategy: data from 2021 to 2022 were used to construct the comorbidity network and extract topological features (the definition of high-cost nodes within the network was also determined entirely based on data from this period); meanwhile, data from 2023 were independently divided into training and testing sets. It should be specifically noted that the network features extracted on this basis are essentially supervised risk representations based on historical data, but no information related to the 2023 data was used during their derivation process. In addition, to eliminate measurement bias caused by macroeconomic fluctuations and inflation, this study uniformly adjusted the total hospitalization costs for 2021 and 2022 to the 2023 value level based on the Consumer Price Index. The complete research flowchart is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Study design and workflow for identifying inpatients with stroke who had high hospitalization costs. This retrospective study used hospital discharge data of inpatients with stroke from a tertiary hospital in Northeast China (2021&#x2010;2023). The workflow consists of three phases: (1) Network construction: data from 2021 to 2022 were used to construct comorbidity networks and extract network features. (2) Feature combination: clinical baselines, conventional comorbidity indices, and the derived network features were integrated for the 2023 cohort. (3) Model development and evaluation: The 2023 dataset was partitioned to train and test 5 machine learning models. Model performances were evaluated, and the Shapley Additive Explanations framework was applied for feature interpretability. Note: a, b, c... represent patient diseases. In the high cost column, &#x201C;1&#x201D; indicates a high-cost patient, and &#x201C;0&#x201D; indicates otherwise. DT: Decision Tree; NN: Neural Network; OR: odds ratio; RF: Random Forest; SVM: Support Vector Machine; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig01.png"/></fig></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was approved by the Scientific Research Ethics Committee of Hangzhou Normal University (approval 2025&#x2010;1026). All procedures strictly complied with the Declaration of Helsinki and relevant ethical guidelines. Data were collected retrospectively through medical record reviews. To ensure patient privacy, all data were deidentified prior to analysis.</p></sec><sec id="s2-3"><title>Network Construction and Feature Extraction</title><p>Prior to network construction, this study established a minimum prevalence threshold, based exclusively on the 2021 to 2022 network construction cohort. Specifically, only diseases with &#x2265;5 affected individuals within this cohort were included as network nodes, thereby ensuring the stability of the constructed network and the statistical representativeness of the extracted features. Additionally, this study incorporated the patients&#x2019; primary and secondary diagnosis codes in the early stage after admission into the network construction process. Considering the extremely high prevalence of the primary diagnosis within the study cohort, to avoid its potential computational bias on the assessment of disease associations, this study performed corresponding statistical corrections in the subsequent calculation formulas for feature indicators.</p></sec><sec id="s2-4"><title>Phenotypic Comorbidity Network</title><p>In this network, nodes represent the included disease categories, and edges represent the pairwise correlations between them. Previous studies have shown that the choice of association measure can significantly affect network topology and research findings [<xref ref-type="bibr" rid="ref26">26</xref>]. Therefore, this study used a co-occurrence correlation as the indicator to quantify the strength of association between disease pairs [<xref ref-type="bibr" rid="ref35">35</xref>]. Compared with traditional measurement methods, the core advantage of this calculation formula lies in its denominator, which uses the arithmetic square root of the sum of the squared occurrence frequencies of each disease. This characteristic makes the indicator more robust when dealing with extremely unbalanced occurrence frequencies of 2 diseases; even if a certain disease is relatively rare, as long as it has a very high co-occurrence ratio with another high-frequency disease, this formula can still objectively assign it a reasonable association weight. To filter out noisy edges caused by coincidental co-occurrences, this study set a minimum edge weight screening mechanism, retaining only associations with an edge weight &#x003E;0.01. The final constructed PCN contained 139 valid nodes and 2404 valid edges, with a network density of 0.251. The specific calculation formula is defined as follows:</p><disp-formula><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>C</mml:mi><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msqrt><mml:mn>2</mml:mn><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:msqrt><mml:msqrt><mml:msubsup><mml:mi>P</mml:mi><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>+</mml:mo><mml:msubsup><mml:mi>P</mml:mi><mml:mi>y</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:msqrt></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <italic>C<sub>xy</sub></italic> refers to the co-occurrence frequency of diseases <italic>x</italic> and <italic>y</italic> within each patient, while <italic>P<sub>x</sub></italic> and <italic>P<sub>y</sub></italic> represent the prevalence rates of diseases <italic>x</italic> and <italic>y</italic>, respectively. <xref ref-type="fig" rid="figure2">Figure 2</xref> illustrates the generated PCN; to optimize visualization, the network displays only edges with weights &#x2265;0.1.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Schematic representation of the phenotype comorbidity network. Node colors denote different <italic>International Classification of Diseases, Tenth Revision</italic> (<italic>ICD-10</italic>) categories. The thickness of the edges indicates the strength of the correlation, and node size represents disease prevalence.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig02.png"/></fig></sec><sec id="s2-5"><title>Distance-Based Disease-Cost Network</title><p>Before constructing the distance-based disease-cost network (DDCN), this study first built an initial network, specifically using the OR to quantify the strength of association between nodes. Unlike the aforementioned network, this network introduced a specific &#x201C;high-cost&#x201D; node in addition to the conventional disease nodes, aiming to further explore the association patterns between specific diseases and high medical expenditures. The calculation formula is defined as follows:</p><disp-formula><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>O</mml:mi><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mo>,</mml:mo><mml:mi>E</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mi>H</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>D</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>D</mml:mi><mml:mi>N</mml:mi></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>E</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>In this framework, <italic>E</italic> represents the specific high-cost outcome. <italic>D<sub>E</sub></italic> and <italic>D<sub>N</sub></italic> represent the number of high-cost and non&#x2013;high-cost patients, respectively, in the patient subgroup with the specific comorbidity <italic>d</italic>. Accordingly, <italic>H<sub>E</sub></italic> and <italic>H<sub>N</sub></italic>, respectively, represent the number of patients corresponding to high-cost and non&#x2013;high-cost states in the control subgroup (ie, patients without the comorbidity <italic>d</italic>). As disease co-occurrence relationships are symmetric, the constructed DDCN is an undirected graph. To address the issue of undefined OR values caused by zero cells in the 2&#x00D7;2 contingency table, this study introduced the Haldane-Anscombe correction, which adds 0.5 to all 4 cells of the contingency table to avoid division-by-zero errors and smooth extreme values generated by small samples. To ensure the high specificity of the network edges, the network only retained significant association edges with an OR &#x003E;1 and a lower bound of its 95% CI &#x003E;1. Subsequently, this study used the min-max normalization formula to transform the ORs into distance metrics, thereby completing the construction of the DDCN. Through this formula, all association strengths were mapped into the standardized distance space of [0, 1], effectively smoothing the extreme numerical fluctuations associated with the extremely high prevalence of the primary diagnosis. The finally constructed DDCN contained 140 valid nodes and 1713 valid edges, with a network density of 0.176.</p><disp-formula><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">D</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">O</mml:mi><mml:mi mathvariant="normal">R</mml:mi></mml:mrow><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula></sec><sec id="s2-6"><title>Feature Engineering and Outcome Variables</title><sec id="s2-6-1"><title>Network Features</title><sec id="s2-6-1-1"><title><italic>Normalized High-Cost Propensity</italic></title><p>Derived from the PCN, this topological feature aims to quantify the cumulative risk associated with high medical expenditures for a specific disease and its adjacent network nodes [<xref ref-type="bibr" rid="ref31">31</xref>,<xref ref-type="bibr" rid="ref35">35</xref>]. Specifically, this indicator effectively captures direct and indirect risk association patterns: if a patient presents with a characteristic disease of the high-cost group, or if their disease exhibits significant topological connectivity with highly prevalent comorbidities in that population, the likelihood of the patient exhibiting a high-cost status increases significantly. The calculation formula for the normalized high-cost propensity (NHCP) of disease <italic>d</italic> is defined as follows:</p><disp-formula><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mi mathvariant="normal">N</mml:mi><mml:mi mathvariant="normal">H</mml:mi><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi mathvariant="normal">&#x03A9;</mml:mi><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>C</mml:mi><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x00A0;</mml:mtext><mml:msub><mml:mi mathvariant="normal">&#x03A9;</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:munder><mml:mi>C</mml:mi><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>d</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>In this calculation formula, <italic>&#x03A9;<sub>d</sub></italic> represents the proportion of high-cost patients among cases with disease <italic>d</italic>, <italic>CC<sub>dc</sub></italic> represents the edge weight between disease <italic>d</italic> and its co-occurring disease c within the PCN, and C constitutes the set of all adjacent nodes directly connected to node <italic>d</italic>. Mechanistically, <italic>&#x03A9;<sub>d</sub></italic> quantifies the inherent and direct high-cost risk associated with disease <italic>d</italic>, while the second term of the formula evaluates the indirect risk brought about by the network spillover effect. By introducing a normalization mechanism through dividing by the denominator <italic>&#x2211;<sub>c&#x2208;C</sub>CC<sub>dc</sub></italic> (ie, the sum of all edge weights connected to <italic>d</italic>), the risk metrics of different nodes are standardized to a uniform scale; this effectively offsets the influence spillover of the primary diagnosis caused by its massive number of connections, enabling the indicator to more objectively reflect the true intensity of comorbidity risks rather than being solely dictated by the absolute number of node connections. As minimum disease prevalence thresholds and edge weight filters were established prior to network construction, certain ICD codes from 2023 might not be incorporated into the network; for such marginal conditions or specific rare diseases, we assumed no additional network connectivity risk and assigned them an NHCP of 0. Ultimately, the NHCP for a patient&#x2019;s diagnostic set <italic>D</italic> is determined as follows:</p><disp-formula><mml:math id="eqn5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">N</mml:mi><mml:mi mathvariant="normal">H</mml:mi><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">N</mml:mi><mml:mi mathvariant="normal">H</mml:mi><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">P</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula></sec><sec id="s2-6-1-2"><title><italic>Shortest Distance</italic></title><p>Shortest distance is a global risk indicator extracted from the DDCN. Distinct from the isolated analysis of a single disease, this indicator uses Dijkstra&#x2019;s algorithm to calculate the weighted shortest path from a specific disease node to the &#x201C;high-cost&#x201D; node, thereby quantifying the topological reachability of a disease leading to a high-cost status via the comorbidity network. The edges fed into the algorithm incorporate both &#x201C;disease-disease&#x201D; and &#x201C;disease-cost&#x201D; associations, allowing diseases that lack direct links to the high-cost node to establish indirect connections via bridging nodes. As shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>, although the direct association between specific disease node <italic>a</italic> and the high-cost node may not reach statistical significance, node <italic>a</italic> exhibits a strong association with node <italic>b</italic>, and node <italic>b</italic> itself is closely related to high medical expenditures. Therefore, node <italic>a</italic> can establish a potential indirect connection to the high-cost status through node <italic>b</italic> acting as a mediating bridge. For isolated nodes lacking an effective connectivity path to the specific &#x201C;high-cost&#x201D; node after screening, this study adopted an extreme-value penalty strategy for assignment: extracting the finite maximum shortest distance length in the current network and adding a constant penalty term (+1) to assign to such isolated nodes. Numerically, this operation maps them to the position furthest from the high-cost node, thereby objectively reflecting their relatively low association strength. Similarly, out-of-network nodes appearing in the 2023 data were also assigned this extreme distance. Ultimately, the shortest distance feature value for the patient&#x2019;s comprehensive diagnostic set <italic>D</italic> is determined as follows:</p><disp-formula><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">S</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mrow><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">r</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">e</mml:mi></mml:mrow><mml:mrow><mml:mi>d</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula></sec></sec></sec><sec id="s2-7"><title>Comorbidity-Related Features</title><p>In addition to the topological features extracted from the comorbidity network, this study also extracted comorbidity-related features similarly derived from patients&#x2019; diagnostic coding data to compare the impact of different feature subsets on the model&#x2019;s identification efficacy in subsequent analyses. Specifically, these features include the number of comorbidities, the CCI, and the ECI.</p></sec><sec id="s2-8"><title>Baseline Features</title><p>This study extracted a total of 8 baseline features from the medical record front pages, specifically including gender, age, insurance type, stroke type, admission route, discharge disposition, LOS, and planned readmission within 31 days. Among these, the insurance type was divided into 4 categories based on the local medical insurance pooling management structure: municipal insurance (applicable to the insured population in the municipal pooling area where the medical institution is located), provincial insurance (corresponding to the insured population in the provincial pooling area), out-of-town insurance (referring to the situation where the insured location and the treatment location belong to different administrative regions), and others (covering various special forms of medical security). The stroke type was defined based on the primary diagnosis code: cases with codes I60 or I61 were classified as hemorrhagic stroke, and cases with code I63 were classified as ischemic stroke. The admission route was divided into emergency, outpatient, and others; the discharge disposition covered routine discharge upon medical advice, discharge against medical advice, and death. The planned readmission within 31 days was dichotomized as &#x201C;yes&#x201D; or &#x201C;no&#x201D; based on the discharge record in the hospital discharge data.</p><p>It is worth noting that among the 8 baseline features mentioned earlier, 5 features&#x2014;gender, age, insurance type, stroke type, and admission way&#x2014;can be obtained at the initial stage of patient admission, whereas discharge disposition, LOS, and planned readmission within 31 days can only be finally determined at the hospital discharge stage, exhibiting an obvious information lag. Accordingly, in the subsequent analysis, we excluded these 3 lagging features and additionally developed a prediction model specifically for patients in the early stage of hospitalization to investigate its performance in the early identification of high-cost patients.</p></sec><sec id="s2-9"><title>Outcome Variables</title><p>Although previous literature has not reached a complete consensus on the definition of high-cost status, using the top 10th percentile as a classification benchmark has been widely recognized and adopted [<xref ref-type="bibr" rid="ref36">36</xref>,<xref ref-type="bibr" rid="ref37">37</xref>]. Consequently, this study defined cases ranking in the top 10% of total hospitalization costs within the study cohort as high-cost patients. Furthermore, the detailed components of inpatient costs were solely used to define the outcome variable and were strictly excluded from subsequent analyses. The selection of this threshold was mainly based on the following considerations: on the one hand, this criterion can effectively alleviate the extreme data imbalance problem associated with setting a threshold too strictly, thereby ensuring the efficacy of subsequent statistical tests and classification evaluations; on the other hand, this criterion can also prevent the dilution of the typical distribution characteristics of the high-cost group caused by a definition scope that is too broad. Furthermore, from the macro perspective of medical policy, the top 10% classification can accurately cover the high-multiplier case groups closely monitored under China&#x2019;s current medical insurance payment system, highly aligning with the practical governance needs of current hospital cost control and refined medical insurance management. To further verify the robustness of the research findings, we conducted expanded testing in the sensitivity analysis using other alternative thresholds.</p></sec><sec id="s2-10"><title>Model Development and Comparison</title><p>This study used the 2023 data subset for the construction and validation of models. To ensure the consistency of the target variable distribution between the training and testing sets, this study adopted the hold-out method, performing a stratified random splitting based on the outcome variable at a ratio of 8:2. After splitting, a training set of 3151 cases (including 315 high-cost cases) and an independent testing set of 787 cases (including 78 high-cost cases) were ultimately generated. Aiming at the class imbalance problem existing between high-cost and non&#x2013;high-cost cases, this study strictly applied the Synthetic Minority Over-sampling Technique (SMOTE) only within the training set before formally training the models. To rigorously prevent overestimation of model performance due to data leakage, no global oversampling was performed prior to cross-validation. Instead, the SMOTE algorithm was specifically embedded within the resampling pipeline. Specifically, during each iteration of the 5-fold cross-validation, SMOTE was dynamically applied only to the training folds, whereas the validation fold always retained the original real-world data distribution. This strategy effectively balanced the class distribution, mitigated the classification bias of machine learning algorithms toward the majority class, and ensured the objectivity of model evaluation. On the basis of the processed data, this study constructed 5 machine learning models to identify high-cost stroke inpatient cases, specifically covering Decision Tree (DT), Support Vector Machine (SVM), Neural Network (NN), Random Forest (RF), and Extreme Gradient Boosting (XGBoost). To guarantee the reproducibility of the experiment and effectively control the risk of overfitting, the model&#x2019;s hyperparameter optimization process used a grid search strategy combined with 5-fold cross-validation strictly executed within the training set. This was done to objectively screen and determine the optimal parameter combination for each model. During the hyperparameter tuning phase, specific search spaces were tailored to the architecture of each model. For DT, the cost-complexity parameter was primarily adjusted. For SVM, both the Radial Basis Function kernel parameter and the regularization penalty were optimized. For NN, the optimization focused on the number of hidden nodes and the weight decay. For RF, the number of candidate variables per split and the minimum node size were fine-tuned. Finally, for XGBoost, a simultaneous optimization was performed on core parameters, including the number of boosting iterations, maximum tree depth, learning rate, and minimum child weight. To balance computational efficiency and mitigate the risk of overfitting, the grid search was restricted to predefined boundaries. Consequently, the identified parameters represent local optima within these ranges rather than exhaustive global optima. The identification performance of all final models was objectively evaluated on the single-split independent hold-out testing set, combined with 1000 Bootstrap resampling to obtain the 95% CIs for various evaluation metrics. To comprehensively measure the overall identification ability of the models, this study adopted the following multidimensional evaluation metric matrix: the area under the curve (AUC), sensitivity, specificity, accuracy, G-mean index, and <italic>F</italic><sub>1</sub>-score [<xref ref-type="bibr" rid="ref38">38</xref>]. Addressing the class imbalance characteristic commonly found in medical data, this study specifically introduced the G-mean and <italic>F</italic><sub>1</sub>-score to more accurately and robustly reflect the overall balanced performance of the model in simultaneously handling the identification of both minority and majority classes. Furthermore, to objectively evaluate the potential gain of network topological features on the model&#x2019;s identification performance, this study used the DeLong test to conduct a statistical examination of the AUC differences between models containing only baseline features and composite models integrating both baseline and network features.</p><p>In addition to the aforementioned evaluation metrics, this study further introduced calibration curves and the Brier score, aiming to evaluate the consistency between the high-cost identification probabilities output by the model and the actual observed true proportions. Finally, this study used decision curve analysis (DCA) to evaluate the net clinical benefit that the model can generate across a wide range of threshold probabilities, thereby verifying its practical application value in medical management and decision-making. Furthermore, to deeply explore the incremental value of the extracted network topological features, this study also implemented an ablation comparative analysis to systematically compare the objective differences in identification performance between &#x201C;models containing only baseline clinical features&#x201D; and &#x201C;models further integrating traditional comorbidity features or network features.&#x201D;</p></sec><sec id="s2-11"><title>Model Interpretability</title><p>To deeply analyze the key associated features behind the high medical expenditures of inpatients with stroke, this study introduced the SHAP technique to provide post-hoc interpretability analysis for the optimal-performing machine learning model. Given that the XGBoost model ultimately used for feature attribution is a tree-based ensemble algorithm, this study specifically selected the TreeExplainer optimized for such algorithms in the analysis to accurately calculate the local and global explanations of features. On the basis of co-operative game theory, the SHAP method can theoretically and fairly distribute the model&#x2019;s identification results among all input variables. Under this framework, the global importance of a feature is quantified by the average level of its absolute SHAP values across the entire study cohort. Specifically, a positive SHAP value indicates that the feature is associated with an increased probability of a case being classified as &#x201C;high cost,&#x201D; while a negative SHAP value denotes a decrease in this classification probability.</p></sec><sec id="s2-12"><title>Statistical Analysis</title><p>This study used R (version 4.4.2; R Core Team) for related statistical analyses and plotting. In the comparative analysis of features between groups, the distribution characteristics of categorical variables were expressed as frequencies and percentages, and the differences between the high-cost and non&#x2013;high-cost groups were compared using the chi-square test; continuous variables underwent a normality test prior to analysis, where data conforming to a normal distribution were presented as mean (SD), and intergroup comparisons were conducted using the independent samples <italic>t</italic> test; continuous variables not conforming to a normal distribution were represented by the median and IQR, and intergroup comparisons adopted the Mann-Whitney <italic>U</italic> test. The visualization of the PCN was completed in Cytoscape (Cytoscape Consortium).</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Patient Characteristics</title><p>According to the previously defined high-cost threshold criteria (the top 10%, equating to &#x2265;94,696.79 RMB), of the 3938 inpatient with stroke incorporated into the 2023 analysis, 393 cases were assigned to the high-cost cohort, and the remaining 3545 cases comprised the non&#x2013;high-cost cohort. <xref ref-type="fig" rid="figure3">Figure 3</xref> displays the distribution of inpatient costs among patients with stroke, with the total health care expenditures of the high-cost group making up 55.1% of the overarching cohort&#x2019;s total costs.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Lorenz curve of inpatient costs in 2023 revealing that the top 10% of inpatients accounted for 55.1% of total costs.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig03.png"/></fig><p>The results of the intergroup analysis presented in <xref ref-type="table" rid="table1">Table 1</xref> reveal that the high-cost and non&#x2013;high-cost groups exhibit statistically significant differences across various features. Specifically, regarding demographics, the proportion of female patients in the high-cost group was significantly higher than that in the non&#x2013;high-cost group; in terms of age distribution, the proportions of patients aged 51 to 60 years and 61 to 70 years in the high-cost group were slightly higher than those in the non&#x2013;high-cost group, whereas the proportion of the population aged over 71 years in the high-cost group was distinctly lower than in the non&#x2013;high-cost group. In terms of clinical features, stroke types exhibited substantial intergroup differences. Ischemic stroke predominated in the overall population, while hemorrhagic stroke accounted for 57% of the high-cost group, a proportion significantly higher than the 14.9% observed in the non&#x2013;high-cost group. Furthermore, the LOS showed a strong positive correlation with high-cost status. The proportion of patients with provincial insurance among high-cost cases was markedly higher than in the non&#x2013;high-cost group. Regarding network features, the intergroup comparison revealed that the NHCP value was significantly elevated in the high-cost group; conversely, the short distance value in this group was significantly reduced. As for comorbidity features, the results indicated that the high-cost patient group had a higher proportion of lower CCI scores and fewer comorbidities compared to the non&#x2013;high-cost group.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Characteristics of patients with stroke with high versus non&#x2013;high hospital costs.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Type and characteristic</td><td align="left" valign="bottom">Non&#x2013;high-cost group (n=3545), n (%)</td><td align="left" valign="bottom">High-cost group (n=393), n (%)</td><td align="left" valign="bottom"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Baseline</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Gender</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">1246 (35.1)</td><td align="left" valign="top">183 (46.6)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">2299 (64.9)</td><td align="left" valign="top">210 (53.4)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2264;50</td><td align="left" valign="top">452 (12.8)</td><td align="left" valign="top">66 (16.8)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>51&#x2010;60</td><td align="left" valign="top">814 (23.0)</td><td align="left" valign="top">110 (28.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>61&#x2010;70</td><td align="left" valign="top">1290 (36.4)</td><td align="left" valign="top">148 (37.7)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;71</td><td align="left" valign="top">989 (27.9)</td><td align="left" valign="top">69 (17.6)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Length of stay</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2264;7</td><td align="left" valign="top">927 (26.1)</td><td align="left" valign="top">64 (16.3)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>8&#x2010;10</td><td align="left" valign="top">1622 (45.8)</td><td align="left" valign="top">64 (16.3)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>11&#x2010;13</td><td align="left" valign="top">607 (17.1)</td><td align="left" valign="top">77 (19.6)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;14</td><td align="left" valign="top">389 (11.0)</td><td align="left" valign="top">188 (47.8)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Stroke type</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">527 (14.9)</td><td align="left" valign="top">224 (57.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">3018 (85.1)</td><td align="left" valign="top">169 (43.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Insurance type</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Municipal</td><td align="left" valign="top">3032 (85.5)</td><td align="left" valign="top">270 (68.7)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Provincial</td><td align="left" valign="top">248 (7.0)</td><td align="left" valign="top">101 (25.7)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Nonlocal</td><td align="left" valign="top">113 (3.2)</td><td align="left" valign="top">10 (2.5)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other</td><td align="left" valign="top">152 (4.3)</td><td align="left" valign="top">12 (3.1)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Admission way</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Outpatient</td><td align="left" valign="top">1403 (39.6)</td><td align="left" valign="top">113 (28.8)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Emergency</td><td align="left" valign="top">2097 (59.2)</td><td align="left" valign="top">278 (70.7)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Other</td><td align="left" valign="top">45 (1.3)</td><td align="left" valign="top">2 (0.5)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Discharge way</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Routine</td><td align="left" valign="top">3269 (92.2)</td><td align="left" valign="top">337 (85.8)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AMA<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></td><td align="left" valign="top">217 (6.1)</td><td align="left" valign="top">30 (7.6)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Death</td><td align="left" valign="top">59 (1.7)</td><td align="left" valign="top">26 (6.6)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Planned 31-day readmission</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">.81</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>No</td><td align="left" valign="top">3461 (97.6)</td><td align="left" valign="top">385 (98.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Yes</td><td align="left" valign="top">84 (2.4)</td><td align="left" valign="top">8 (2.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">Comorbidity features</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>CCI<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0</td><td align="left" valign="top">2072 (58.4)</td><td align="left" valign="top">291 (74.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>1</td><td align="left" valign="top">1211 (34.2)</td><td align="left" valign="top">86 (21.9)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;2</td><td align="left" valign="top">262 (7.4)</td><td align="left" valign="top">16 (4.1)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>ECI<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup></td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">0.24</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2264;0</td><td align="left" valign="top">2916 (82.3)</td><td align="left" valign="top">310 (78.9)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2&#x2010;5</td><td align="left" valign="top">414 (11.7)</td><td align="left" valign="top">56 (14.2)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;6</td><td align="left" valign="top">215 (6.1)</td><td align="left" valign="top">27 (6.9)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Number of comorbidities</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x2003;</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>0&#x2010;1</td><td align="left" valign="top">1628 (45.9)</td><td align="left" valign="top">289 (73.5)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>2&#x2010;3</td><td align="left" valign="top">1423 (40.1)</td><td align="left" valign="top">59 (15.0)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>&#x2265;4</td><td align="left" valign="top">494 (13.9)</td><td align="left" valign="top">45 (11.5)</td><td align="left" valign="top">&#x2003;</td></tr><tr><td align="left" valign="top">Network features</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>NHCP<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup>, median (IQR)</td><td align="left" valign="top">0.07 (0.07-0.13)</td><td align="left" valign="top">0.27 (0.06-0.32)</td><td align="left" valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Shortest distance, median (IQR)</td><td align="left" valign="top">2.00 (1.97-2.00)</td><td align="left" valign="top">1.00 (0.99-2.89)</td><td align="left" valign="top">&#x003C;.001</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AMA: against medical advice.</p></fn><fn id="table1fn2"><p><sup>b</sup>CCI: Charlson Comorbidity Index.</p></fn><fn id="table1fn3"><p><sup>c</sup>ECI: Elixhauser Comorbidity Index.</p></fn><fn id="table1fn4"><p><sup>d</sup>NHCP: normalized high-cost propensity.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Model Development and Performance Comparison</title><p><xref ref-type="table" rid="table2">Table 2</xref> presents the identification performance of the 5 machine learning models after incorporating the 2 network features. Overall, after incorporating the network-derived features, the identification performance of all models achieved a certain degree of improvement. Among them, XGBoost exhibited the largest AUC improvement, increasing from 0.814 to 0.899; moreover, the remaining evaluation metrics of this model also demonstrated the optimal performance among the 5 models.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Performance comparison of 5 machine learning models based on baseline and network-augmented feature sets. The data in parentheses are 95% CI.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Model</td><td align="left" valign="bottom" colspan="2">Feature group</td></tr><tr><td align="left" valign="bottom">Baseline</td><td align="left" valign="bottom">Baseline<italic>+</italic>network</td></tr></thead><tbody><tr><td align="left" valign="top" colspan="3">DT<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AUC<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.824 (0.771&#x2010;0.870)</td><td align="left" valign="top">0.878 (0.835&#x2010;0.916)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.766 (0.654&#x2010;0.926)</td><td align="left" valign="top">0.893 (0.815&#x2010;0.963)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.803 (0.631&#x2010;0.868)</td><td align="left" valign="top">0.767 (0.675&#x2010;0.816)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.799 (0.656&#x2010;0.856)</td><td align="left" valign="top">0.779 (0.703&#x2010;0.822)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-mean</td><td align="left" valign="top">0.780 (0.731&#x2010;0.830)</td><td align="left" valign="top">0.827 (0.786&#x2010;0.863)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.439 (0.315&#x2010;0.532)</td><td align="left" valign="top">0.445 (0.366&#x2010;0.520)</td></tr><tr><td align="left" valign="top" colspan="3">SVM<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AUC</td><td align="left" valign="top">0.829 (0.780&#x2010;0.876)</td><td align="left" valign="top">0.861 (0.809&#x2010;0.909)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.826 (0.742&#x2010;0.907)</td><td align="left" valign="top">0.810 (0.709&#x2010;0.921)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.792 (0.753&#x2010;0.826)</td><td align="left" valign="top">0.829 (0.701&#x2010;0.882)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.795 (0.759&#x2010;0.827)</td><td align="left" valign="top">0.827 (0.718&#x2010;0.873)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-mean</td><td align="left" valign="top">0.808 (0.764&#x2010;0.850)</td><td align="left" valign="top">0.818 (0.773&#x2010;0.863)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.443 (0.369&#x2010;0.513)</td><td align="left" valign="top">0.483 (0.372&#x2010;0.567)</td></tr><tr><td align="left" valign="top" colspan="3">NN<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AUC</td><td align="left" valign="top">0.827 (0.774&#x2010;0.870)</td><td align="left" valign="top">0.876 (0.832&#x2010;0.916)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.804 (0.687&#x2010;0.927)</td><td align="left" valign="top">0.843 (0.744&#x2010;0.950)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.773 (0.611&#x2010;0.857)</td><td align="left" valign="top">0.818 (0.683&#x2010;0.879)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.776 (0.642&#x2010;0.848)</td><td align="left" valign="top">0.820 (0.708&#x2010;0.873)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-mean</td><td align="left" valign="top">0.787 (0.738&#x2010;0.831)</td><td align="left" valign="top">0.829 (0.783&#x2010;0.870)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.419 (0.324&#x2010;0.508)</td><td align="left" valign="top">0.486 (0.378&#x2010;0.577)</td></tr><tr><td align="left" valign="top" colspan="3">RF<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AUC</td><td align="left" valign="top">0.823 (0.768&#x2010;0.869)</td><td align="left" valign="top">0.891 (0.848&#x2010;0.929)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.822 (0.671&#x2010;0.940)</td><td align="left" valign="top">0.824 (0.732&#x2010;0.912)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.747 (0.608&#x2010;0.882)</td><td align="left" valign="top">0.858 (0.774&#x2010;0.898)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.755 (0.639&#x2010;0.864)</td><td align="left" valign="top">0.855 (0.787&#x2010;0.889)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-mean</td><td align="left" valign="top">0.782 (0.734&#x2010;0.824)</td><td align="left" valign="top">0.840 (0.796&#x2010;0.886)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.402 (0.312&#x2010;0.500)</td><td align="left" valign="top">0.530 (0.425&#x2010;0.606)</td></tr><tr><td align="left" valign="top" colspan="3">XGBoost<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>AUC</td><td align="left" valign="top">0.814 (0.758&#x2010;0.863)</td><td align="left" valign="top">0.899 (0.857&#x2010;0.936)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sensitivity</td><td align="left" valign="top">0.784 (0.630&#x2010;0.905)</td><td align="left" valign="top">0.826 (0.733&#x2010;0.949)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Specificity</td><td align="left" valign="top">0.770 (0.681&#x2010;0.881)</td><td align="left" valign="top">0.865 (0.698&#x2010;0.908)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Accuracy</td><td align="left" valign="top">0.771 (0.696&#x2010;0.864)</td><td align="left" valign="top">0.861 (0.723&#x2010;0.900)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>G-mean</td><td align="left" valign="top">0.774 (0.725&#x2010;0.818)</td><td align="left" valign="top">0.844 (0.798&#x2010;0.891)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.412 (0.319&#x2010;0.522)</td><td align="left" valign="top">0.546 (0.394&#x2010;0.635)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>DT: Decision Tree.</p></fn><fn id="table2fn2"><p><sup>b</sup>AUC: area under the curve. </p></fn><fn id="table2fn3"><p><sup>c</sup>SVM: Support Vector Machine.</p></fn><fn id="table2fn4"><p><sup>d</sup>NN: Neural Network.</p></fn><fn id="table2fn5"><p><sup>e</sup>RF: random forest.</p></fn><fn id="table2fn6"><p><sup>f</sup>XGBoost: Extreme Gradient Boosting.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="fig" rid="figure4">Figure 4</xref> displays the receiver operating characteristic curves of the 5 machine learning models combined with network features; the DeLong test results in <xref ref-type="fig" rid="figure4">Figure 4B</xref> further confirmed that the difference in AUC gain between the &#x201C;composite model fusing network features&#x201D; and the &#x201C;initial model containing only baseline features&#x201D; is statistically significant. Specifically, the difference in AUC between the 2 models was 0.085 (95% CI 0.052&#x2010;0.123), with an exact <italic>P</italic> value of 1.42&#x00D7;10&#x207B;&#x2076; according to the DeLong test.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Receiver operating characteristic (ROC) curves of different machine learning models for identifying high-cost patients and comparison of feature gains. AUC: area under the curve; DT: Decision Tree; NN: Neural Network; OR: odds ratio; RF: Random Forest; SVM: Support Vector Machine; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig04.png"/></fig><p>The calibration curve results in <xref ref-type="fig" rid="figure5">Figure 5A</xref> show that the curve trajectory of the XGBoost model is closest to the ideal reference line, indicating a high degree of consistency between the high-cost identification probabilities output by the model and the actual observed true proportions. Simultaneously, XGBoost achieved a Brier score of 0.087, belonging to the low-error tier alongside RF. Its overall calibration performance was significantly better than that of the SVM, DT, and NN, further confirming the robust reliability of its output identification probability estimates. In the DCA shown in <xref ref-type="fig" rid="figure5">Figure 5B</xref>, XGBoost similarly highlighted significant practical application advantages. Specifically, within the broad threshold probability interval of clinical and management significance, the net benefit curve of XGBoost consistently remained at the highest level; this means that compared to other machine learning models participating in the evaluation, using this model to guide medical resource allocation can yield a more substantial comprehensive net benefit. Given the outstanding comprehensive identification performance described earlier, this study ultimately selected XGBoost as the core model for subsequent feature attribution and mechanism analysis. The optimal hyperparameter combinations and their specific configurations for each model are detailed in Table A1 in the <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Evaluation of calibration and clinical utility of various machine learning models. AUC: area under the curve; DT: Decision Tree; NN: Neural Network; OR: odds ratio; RF: Random Forest; SVM: Support Vector Machine; XGBoost: Extreme Gradient Boosting.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig05.png"/></fig></sec><sec id="s3-3"><title>Ablation Analysis</title><p>To objectively evaluate the specific contributions of traditional comorbidity features and single network topological features to the model&#x2019;s identification performance, this study further implemented an ablation analysis relying on the XGBoost model. The outcomes are presented in <xref ref-type="table" rid="table3">Table 3</xref>: in terms of traditional comorbidity features, regardless of the introduction of CCI, ECI, or a basic comorbidity count, the model&#x2019;s identification performance demonstrated a certain degree of enhancement; however, its overall performance still fell markedly short of the model integrating network features. In contrast, the incorporation of CCI led to a certain degree of performance degradation. Furthermore, the 2 network topological features also had different focuses regarding their gains in identification performance. Specifically, the &#x201C;baseline+SD&#x201D; model demonstrated better discrimination in the AUC metric; in contrast, although the &#x201C;baseline+NHCP&#x201D; model had a slightly lower AUC value, it performed better on the <italic>F</italic><sub>1</sub>-score. Finally, by rigorously excluding lagging variables that could only be determined after discharge and retaining only features available in the early stage of hospitalization, we independently developed a corresponding early identification model. The results showed that, although the model achieved an acceptable AUC, its <italic>F</italic><sub>1</sub>-score was markedly low, suggesting a potential imbalance in model performance when based solely on the currently available features.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Performance evaluation of models based on single network features, traditional comorbidity features, and preadmission features. The data in parentheses are 95% CI.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Feature group</td><td align="left" valign="bottom" colspan="6">Performance</td></tr><tr><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup></td><td align="left" valign="bottom">Sensitivity</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">Baseline+CCI<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="top">0.807 (0.747&#x2010;0.863)</td><td align="left" valign="top">0.746 (0.608&#x2010;0.867)</td><td align="left" valign="top">0.801 (0.679&#x2010;0.905)</td><td align="left" valign="top">0.796 (0.695&#x2010;0.879)</td><td align="left" valign="top">0.771 (0.714&#x2010;0.819)</td><td align="left" valign="top">0.423 (0.329&#x2010;0.532)</td></tr><tr><td align="left" valign="top">Baseline+ECI<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">0.827 (0.773&#x2010;0.874)</td><td align="left" valign="top">0.812 (0.714&#x2010;0.899)</td><td align="left" valign="top">0.774 (0.687&#x2010;0.829)</td><td align="left" valign="top">0.777 (0.702&#x2010;0.823)</td><td align="left" valign="top">0.792 (0.742&#x2010;0.836)</td><td align="left" valign="top">0.420 (0.338&#x2010;0.491)</td></tr><tr><td align="left" valign="top">Baseline+comorbidity count</td><td align="left" valign="top">0.852 (0.808&#x2010;0.891)</td><td align="left" valign="top">0.804 (0.690&#x2010;0.926)</td><td align="left" valign="top">0.785 (0.670&#x2010;0.864)</td><td align="left" valign="top">0.787 (0.693&#x2010;0.853)</td><td align="left" valign="top">0.793 (0.748&#x2010;0.839)</td><td align="left" valign="top">0.432 (0.336&#x2010;0.519)</td></tr><tr><td align="left" valign="top">Baseline+NHCP<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">0.888 (0.845&#x2010;0.925)</td><td align="left" valign="top">0.832 (0.743&#x2010;0.910)</td><td align="left" valign="top">0.843 (0.802&#x2010;0.883)</td><td align="left" valign="top">0.842 (0.804&#x2010;0.878)</td><td align="left" valign="top">0.837 (0.793&#x2010;0.880)</td><td align="left" valign="top">0.511 (0.435&#x2010;0.591)</td></tr><tr><td align="left" valign="top">Baseline+shortest distance</td><td align="left" valign="top">0.894 (0.853&#x2010;0.929)</td><td align="left" valign="top">0.876 (0.756&#x2010;0.972)</td><td align="left" valign="top">0.797 (0.694&#x2010;0.893)</td><td align="left" valign="top">0.804 (0.717&#x2010;0.883)</td><td align="left" valign="top">0.833 (0.793&#x2010;0.874)</td><td align="left" valign="top">0.477 (0.372&#x2010;0.592)</td></tr><tr><td align="left" valign="top">Baseline-A<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup>+network</td><td align="left" valign="top">0.848 (0.803&#x2010;0.890)</td><td align="left" valign="top">0.869 (0.759&#x2010;0.942)</td><td align="left" valign="top">0.718 (0.660&#x2010;0.808)</td><td align="left" valign="top">0.733 (0.681&#x2010;0.808)</td><td align="left" valign="top">0.789 (0.750&#x2010;0.829)</td><td align="left" valign="top">0.393 (0.321&#x2010;0.468)</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>AUC: area under the curve.</p></fn><fn id="table3fn2"><p><sup>b</sup>CCI: Charlson Comorbidity Index.</p></fn><fn id="table3fn3"><p><sup>c</sup>ECI: Elixhauser Comorbidity Index.</p></fn><fn id="table3fn4"><p><sup>d</sup>NHCP: normalized high-cost propensity.</p></fn><fn id="table3fn5"><p><sup>e</sup>Baseline features available at admission.</p></fn></table-wrap-foot></table-wrap><p>Previous intergroup analysis of baseline features showed a statistically significant difference in stroke type between the 2 groups, indicating a high degree of clinical relevance between this feature and high-cost status. Given that all diagnostic codes, including the principal diagnosis, were incorporated in the construction of the comorbidity network, the resulting network topological features to some extent captured information inherent to stroke subtypes. To deeply explore the potential multicollinearity and information overlap effects among features, this study additionally constructed a model that excluded the stroke subtype and retained only the network features. The results showed that after excluding this subtype feature, the model&#x2019;s identification performance was basically unaffected (relevant evaluation data are detailed in Table A2 in the <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>). In other words, the value of network features lies more in complementing existing clinical information than in providing discriminative ability that is entirely independent of stroke subtypes. Therefore, the incremental value of network features should be interpreted as a marginal improvement beyond established clinical variables, rather than as a complete replacement for conventional features. Table A2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> also summarizes the performance comparison of models fusing all feature combinations. The comprehensive analytical results corroborated that the feature combination of &#x201C;baseline characteristics+dual network features&#x201D; delivered the most optimal overall identification capabilities across all evaluation metrics. Therefore, this study ultimately selected this feature subset as the standard input for the subsequent SHAP feature attribution analysis.</p><p>Meanwhile, to verify the applicability and robustness of the aforementioned identification framework in specific clinical subgroups, this study further evaluated the model&#x2019;s performance within the hemorrhagic and ischemic stroke subgroups in the independent testing set, based on this optimal feature combination. The stratified analysis results (<xref ref-type="table" rid="table4">Table 4</xref>) showed that although the AUC value of each model within a single subgroup slightly decreased by approximately 0.03 to 0.04 compared to the overall cohort, they still demonstrated stable identification performance across different clinical subgroups.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Subgroup analysis of machine learning model identification performance across different stroke types.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model and subgroup</td><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup> (95% CI)</td></tr></thead><tbody><tr><td align="left" valign="top">DT<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">0.828 (0.769&#x2010;0.877)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">0.830 (0.770&#x2010;0.888)</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">0.780 (0.690&#x2010;0.860)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">0.821 (0.740&#x2010;0.895)</td></tr><tr><td align="left" valign="top">NN<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">0.790 (0.707&#x2010;0.864)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">0.842 (0.767&#x2010;0.905)</td></tr><tr><td align="left" valign="top">RF<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">0.849 (0.775&#x2010;0.913)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">0.848 (0.776&#x2010;0.907)</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Hemorrhagic</td><td align="left" valign="top">0.868 (0.798&#x2010;0.927)</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Ischemic</td><td align="left" valign="top">0.856 (0.783&#x2010;0.915)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>AUC: area under the curve. </p></fn><fn id="table4fn2"><p><sup>b</sup>DT: Decision Tree.</p></fn><fn id="table4fn3"><p><sup>c</sup>SVM: Support Vector Machine.</p></fn><fn id="table4fn4"><p><sup>d</sup>NN: Neural Network.</p></fn><fn id="table4fn5"><p><sup>e</sup>RF: Random Forest.</p></fn><fn id="table4fn6"><p><sup>f</sup>XGBoost: Extreme Gradient Boosting.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>Model Interpretability</title><p>To enhance the transparency and interpretability of the model in the context of clinical decision-making, this study introduced the SHAP method to objectively quantify the marginal contribution of each input variable to the final identification result. This method covers 2 dimensions: global interpretability at the feature level and local interpretability at the individual level. In terms of global interpretability, the bar chart in <xref ref-type="fig" rid="figure6">Figure 6A</xref> is sorted in descending order based on the mean absolute SHAP values of the features, intuitively illustrating the relative importance of different variables in the identification of high-cost patients. The analysis results showed that the top 5 core associated features contributing to the model&#x2019;s performance were, in order, short distance, LOS, NHCP, age, and insurance type. As shown in the pie chart in <xref ref-type="fig" rid="figure6">Figure 6A</xref>, the cumulative contribution of network features accounts for 50.4% of the model&#x2019;s total output. Among the baseline features, baseline features available at admission account for 19.7%, while baseline features available at hospital discharge reach 29.9%. The SHAP beeswarm plot in <xref ref-type="fig" rid="figure6">Figure 6B</xref> intuitively presents the distribution trend of SHAP values across various features for the entire sample. <xref ref-type="fig" rid="figure6">Figure 6C</xref> displays the SHAP dependence plots for the top 3 features, showing how these variables regulate the model&#x2019;s output within different value ranges. Specifically, a lower value of shortest distance, along with higher values of LOS and NHCP, is significantly associated with a high probability of cases being identified as a high-cost status. The SHAP interaction analysis in <xref ref-type="fig" rid="figure6">Figure 6D</xref> reveals that the interaction effects between network features and LOS exhibit complex nonlinear synergistic associations. Specifically, patients with a LOS &#x2265;14 days show a significant inverted U-shaped trend, with the interaction value reaching a positive peak when shortest distance is at a moderate level, significantly elevating the high-cost risk; the 11&#x2010; to 13-day group displays a U-shape with mostly negative interactions; whereas the &#x2264;10-day group has a flat interaction effect, fluctuating around the baseline. On the other hand, when examining the interaction between NHCP and LOS: for patients with an LOS of 8 to 10 days, the interaction effect shows an upward trend, turning from a negative value to a positive value as NHCP increases. Conversely, for the subgroup with an LOS &#x2265;14 days, the interaction effect presents a downward trend, turning negative with the increase of NHCP. In contrast, for the subgroups with an LOS &#x2264;7 days and 11 to 13 days, the interaction values remain relatively stable across different levels of NHCP.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Global interpretation of the Extreme Gradient Boosting model using Shapley Additive Explanations (SHAP) analysis. (A) SHAP feature importance bar chart, Baseline-A: Baseline features available at admission; Baseline-H: Baseline features available at hospital discharge. (B) SHAP summary plot (beeswarm) showing the distribution of feature impacts. Each dot represents a sample; color indicates feature value (yellow=high and purple=low). Taking StayDays as an example, yellow dots (longer stays) are predominantly distributed on the positive SHAP side, indicating that prolonged hospitalization increases the likelihood of high-cost classification, while purple dots (shorter stays) cluster on the negative side, suggesting an inhibitory effect on high-cost generation. (C) SHAP dependence plots for top features (SD, StayDays, and normalized high-cost propensity [NHCP]), illustrating the marginal relationship between the feature value and its impact on the model output. (D) SHAP interaction plots illustrating the interaction effects of SD and NHCP with length of stay. Variable name mapping: StayDays=Length of Stay; InsType=Insurance Type; AgeGroup=Age; AdmitWay=Admission Way; DischWay=Discharge Way; ReAdm31d=Planned 31-day Readmission.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig06.png"/></fig><p>At the local explanation level, we generated SHAP force plots to elucidate the contribution of each feature to the identification result for a specific individual. <xref ref-type="fig" rid="figure7">Figure 7A</xref> displays a patient who was identified by the model as highly likely to be in a high-cost status. The core features making the primary positive contribution to this identification probability were shortest distance and NHCP, while the LOS played a certain mitigating role. Conversely, the patient in <xref ref-type="fig" rid="figure7">Figure 7B</xref> was identified as having an extremely low probability of being in a high-cost status. The dominant associated features supporting the identification of this case into the non&#x2013;high-cost group were, in order, a higher shortest distance, a shorter LOS, a lower NHCP, and a specific insurance type.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Local explanations for individual predictions using Shapley Additive Explanations force plots. The plots illustrate how specific feature values contribute to pushing the model&#x2019;s output from the base value (E[f(x)]) toward the final prediction (f(x)). Yellow bars indicate features that increase the likelihood of being a high-cost patient (positive contribution), while purple bars indicate features that decrease it (negative contribution). (A) A representative high-cost patient sample (f(x)=0.894). (B) A representative non&#x2013;high-cost patient sample (f(x)=&#x2212;4.740). Variable name mapping: StayDays=Length of Stay; InsType=Insurance Type; AgeGroup=Age. NHCP: normalized high-cost propensity.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e93680_fig07.png"/></fig></sec><sec id="s3-5"><title>Sensitivity Analysis</title><p>As presented in <xref ref-type="table" rid="table5">Table 5</xref>, at the extreme threshold of 5%, the model&#x2019;s AUC value reached its highest level; however, affected by the extremely imbalanced data, the model&#x2019;s <italic>F</italic><sub>1</sub>-score was below 0.4, meaning that the model performance was poor and the identification ability was relatively weak. As the threshold was relaxed, although the model&#x2019;s AUC value experienced a slight decline, it showed a relatively stable trend within the 10% to 20% range. In addition, due to the steady increase in the number of positive samples, the class imbalance phenomenon was effectively alleviated, prompting the <italic>F</italic><sub>1</sub>-scores of all models to show a gradually increasing trend. Among them, the threshold adjustment from 5% to 10% optimized the model performance most significantly, and the <italic>F</italic><sub>1</sub>-scores of all models achieved substantial improvements. This indicates that the 10% threshold effectively improved the problem of impaired model performance caused by extreme class imbalance under the low threshold.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Sensitivity analysis of model performance under varying thresholds. The data in parentheses are 95% CI.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom" rowspan="2">Thresholds</td><td align="left" valign="bottom" colspan="6">Performance</td></tr><tr><td align="left" valign="bottom">AUC<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup></td><td align="left" valign="bottom">Sensitivity</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">5%</td><td align="left" valign="top">0.939 (0.914&#x2010;0.963)</td><td align="left" valign="top">0.951 (0.881&#x2010;1.000)</td><td align="left" valign="top">0.847 (0.814&#x2010;0.880)</td><td align="left" valign="top">0.852 (0.820&#x2010;0.884)</td><td align="left" valign="top">0.897 (0.859&#x2010;0.930)</td><td align="left" valign="top">0.389 (0.296&#x2010;0.479)</td></tr><tr><td align="left" valign="top">10%</td><td align="left" valign="top">0.899 (0.857&#x2010;0.936)</td><td align="left" valign="top">0.826 (0.733&#x2010;0.949)</td><td align="left" valign="top">0.865 (0.698&#x2010;0.908)</td><td align="left" valign="top">0.861 (0.723&#x2010;0.900)</td><td align="left" valign="top">0.844 (0.798&#x2010;0.891)</td><td align="left" valign="top">0.546 (0.394&#x2010;0.635)</td></tr><tr><td align="left" valign="top">15%</td><td align="left" valign="top">0.906 (0.875&#x2010;0.933)</td><td align="left" valign="top">0.884 (0.760&#x2010;0.966)</td><td align="left" valign="top">0.806 (0.720&#x2010;0.914)</td><td align="left" valign="top">0.818 (0.748&#x2010;0.895)</td><td align="left" valign="top">0.842 (0.811&#x2010;0.872)</td><td align="left" valign="top">0.597 (0.508&#x2010;0.700)</td></tr><tr><td align="left" valign="top">20%</td><td align="left" valign="top">0.898 (0.869&#x2010;0.923)</td><td align="left" valign="top">0.894 (0.821&#x2010;0.950)</td><td align="left" valign="top">0.789 (0.728&#x2010;0.857)</td><td align="left" valign="top">0.810 (0.767&#x2010;0.858)</td><td align="left" valign="top">0.839 (0.812&#x2010;0.868)</td><td align="left" valign="top">0.653 (0.591&#x2010;0.712)</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>AUC: area under the curve.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>Unlike most previous models, which incorporated clinical laboratory, imaging, or genomic data [<xref ref-type="bibr" rid="ref39">39</xref>-<xref ref-type="bibr" rid="ref41">41</xref>], this study used hospital discharge data to develop a high-performance model for identifying patients with stroke who have high hospitalization costs by combining diagnostic network analysis with machine learning algorithms. Compared with these data sources, hospital discharge data offer several important advantages, including greater standardization, lower cost, and broad availability across hospitals at all levels, making it a practical basis for implementation in diverse health care settings. Although this data source has inherent limitations in dimensionality, incorporating topological features derived from the comorbidity network substantially improved the performance of all models, consistent with previous studies [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Moreover, compared with other models for identifying high-need, high-cost patients that did not include network features, our model showed superior performance [<xref ref-type="bibr" rid="ref17">17</xref>,<xref ref-type="bibr" rid="ref42">42</xref>-<xref ref-type="bibr" rid="ref44">44</xref>].</p><p>Our results indicate that, although both are derived from diagnostic data, network features from the comorbidity network contributed more to model performance than conventional comorbidity measures such as CCI, ECI, and comorbidity count. By quantifying the pathway from a specific disease node to high-cost outcomes and integrating the disease&#x2019;s intrinsic risk with the combined effects of its local neighborhood, these features captured latent information that is difficult to detect using traditional statistical methods, thereby substantially improving the model&#x2019;s ability to identify high-cost patients. Additionally, subgroup analyses across different stroke subtypes further verify the stability and reliability of the comorbidity network features extracted in this study. The model maintained satisfactory predictive performance under the macroscopic classification of hemorrhagic and ischemic stroke, demonstrating good applicability across major clinical subgroups. Nevertheless, this classification is relatively coarse and ignores substantial clinical heterogeneity among different subtypes within hemorrhagic stroke. We therefore conducted a further refined subgroup analysis focusing on subarachnoid hemorrhage (I60) and intracerebral hemorrhage (I61), as presented in Table A3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The results showed that the model performed stably in the I61 subgroup but yielded unsatisfactory outcomes in the I60 subgroup. This discrepancy can be explained by the sample characteristics of the I60 population. This subgroup had a small sample size and a high degree of internal homogeneity. Among the total of 247 patients with I60, 170 were categorized as high-cost cases, indicating highly consistent hospitalization costs in this subtype. Limited information from routine hospital discharge data failed to capture subtle individual differences, which impaired the model&#x2019;s discriminative ability. In future research, we will expand the sample size and incorporate more clinical indicators, treatment-related information, and other variables to develop refined models for individual stroke subtypes.</p><p>In the global feature importance ranking, LOS was the second most important feature, after shortest distance. Previous studies have likewise highlighted the critical role of LOS in identifying high-cost pediatric inpatients [<xref ref-type="bibr" rid="ref42">42</xref>]. SHAP interaction analysis further showed that the association between LOS and comorbidity network features was strongly time dependent, suggesting that the trajectory of medical resource use in stroke may evolve dynamically over the course of the disease. In the subgroup with an LOS of 8 to 10 days, higher NHCP showed a clear positive synergistic interaction, which may reflect the clinical profile of complex cases requiring highly resource-intensive interventions during the acute phase. However, when LOS was &#x2265;14 days, high NHCP instead showed a strong negative interaction; at the same time, this long-stay subgroup exhibited a positive interaction peak at moderate levels of shortest distance. This time-varying pattern may suggest a law of diminishing marginal costs: as hospitalization becomes substantially prolonged, the main drivers of high expenditure may shift from fluctuations in acute disease severity to the cumulative baseline costs of ongoing care. Meanwhile, the 11 to 13 days subgroup showed a U-shaped pattern, with negative interactions across most of the value range, which may indicate a transitional stage in cost generation as patients move from the acute phase to the longer-term recovery phase.</p><p>Age and insurance type were also important determinants of the model&#x2019;s identification performance. In particular, the effect of age on high hospitalization costs showed a nonlinear pattern, with older age exerting a negative effect on the identification of high-cost patients. One possible explanation is that because comorbid degenerative conditions increase surgical risk [<xref ref-type="bibr" rid="ref45">45</xref>], older patients are more likely to receive conservative treatment and, compared with younger patients, are less likely to undergo high-cost interventional procedures, such as mechanical thrombectomy [<xref ref-type="bibr" rid="ref46">46</xref>]. This pattern of clinical management is consistent with the baseline finding that patients in the high-cost group were, on average, younger. In addition, this study found a relatively high proportion of patients with a low comorbidity burden in the high-cost group. As patients in this group were generally younger, the prevalence of chronic underlying diseases would be expected to be lower, which may partly explain the clustering of relatively mild comorbidity in this group. However, given the inherent limitations of retrospective observational studies, future research should further examine the causal relationships among age, comorbidity, and high hospitalization costs in patients with stroke through prospective study designs. Notably, the performance of CCI, comorbidity count, and ECI in between-group comparisons was not entirely consistent, which may be attributable to differences in disease composition and weighting structure among these comorbidity measures. Comorbidity count reflects only the number of coexisting conditions and does not incorporate disease-specific weights. In contrast, the CCI applies weights based on a relatively limited set of comorbidity categories, whereas the ECI encompasses a broader range of comorbidity conditions and is based on a different scoring framework. Regarding insurance type, patients covered by provincial medical insurance generally benefit from higher reimbursement rates and broader formulary coverage. As a result, they may have greater financial access to high-cost diagnostic and therapeutic services and, when treated at tertiary grade A hospitals, may be more likely to accept more comprehensive or resource-intensive interventions. This may explain the significant positive association between provincial medical insurance and high-cost status. In contrast, patients with cross-provincial medical insurance or other payment methods may be constrained by reimbursement limits or out-of-pocket affordability. In clinical decision-making, these patients may therefore be more likely to adopt more economical or conservative treatment strategies, which is reflected in the model as a lower probability of high medical expenditure. Clinical management in these groups may prioritize cost-effective care, resulting in a negative association with high-cost status. This observed pattern is consistent with the findings of Yang et al in the Chinese ischemic stroke population [<xref ref-type="bibr" rid="ref47">47</xref>].</p><p>Sensitivity analysis showed that when the outcome threshold was set at 5%, the constructed model performed poorly, with limited practical value for management purposes. As the threshold was relaxed from 10% to 20%, the model&#x2019;s AUC tended to stabilize, while the <italic>F</italic><sub>1</sub>-score increased. However, this improvement in <italic>F</italic><sub>1</sub>-score may mainly reflect the passive effect of a larger number of positive samples and a more balanced class distribution, rather than a true improvement in the model&#x2019;s discriminative ability. In addition, a broader threshold of 15% or 20% may dilute the core characteristics of the truly high-expenditure group, thereby creating potential challenges for the precise allocation of clinical resources and the refined management of medical insurance cost control. In contrast, a 10% threshold allows greater concentration on the core high-cost group and more accurately targets the disproportionately high-expenditure cases prioritized under the current medical insurance payment system. It should be noted, however, that at this 10% prevalence level, the optimal model yields a positive predictive value of approximately 0.41. While this implies a certain false-positive fraction, these misclassified cases typically represent moderately complex patients who may still benefit from subsequent care management. Ultimately, the practical implementation of this model relies on available administrative capacity. If hospital resources are relatively sufficient, health care managers can effectively absorb the interventions for these false positives and may even appropriately increase the threshold to expand coverage, thereby minimizing the risk of overlooking other potential high-cost cases.</p></sec><sec id="s4-2"><title>Limitations and Future Directions</title><p>First, the training and test sets in this study were generated using a single stratified random split, and the model evaluation results may therefore be influenced by variability arising from random sample partitioning. Although bootstrap resampling was used to estimate CIs, it could not fully eliminate the bias inherent to a single split. Due to the constraints of the study design for temporal validation and to preserve the original temporal characteristics of the cohort, repeated data splitting was not performed. Future studies could further enhance the reliability of the results through repeated split-sample validation. Second, the data were primarily sourced from a single center, which may introduce biases related to region-specific medical practices or insurance policies. Future research should prioritize multicenter external validation to confirm the model&#x2019;s robustness; furthermore, the generalizability of this network feature-based framework to other disease spectra remains to be validated. Third, given the retrospective nature of this study and its reliance on hospital discharge data, some features exhibit information lag, which limits the utility of the model for preadmission screening. Although we used features available at admission to construct an early identification model, achieving more precise screening will require incorporating other features available at admission based on this study. However, SHAP-based interpretability analysis still holds significant value in providing decision support for clinical process management and guiding personalized interventions. Additionally, a specific limitation exists regarding the &#x201C;planned readmission within 31 days&#x201D; indicator. Owing to our 30-day aggregation strategy, merged readmission episodes accrue higher total costs, introducing a potential circularity issue that could theoretically inflate this feature&#x2019;s apparent contribution. Although our SHAP analysis revealed it had the lowest importance contribution&#x2014;indicating a minimal practical impact on the model&#x2019;s identification performance in this cohort&#x2014;future studies should carefully account for this potential confounding effect when constructing cost-related variables. Fourth, although the combination of comorbidity network features enriches the feature space, the current absence of biomarkers may limit the granularity of disease severity representation. Future research should prioritize the development of multimodal fusion models to evaluate the incremental value of biomarkers relative to network topological features, thereby establishing a more comprehensive and multidimensional risk profile for patient expenditures. Fifth, the static network constructed in this study may not fully capture high-order dependencies or the dynamic trajectories of disease evolution. Therefore, future research should integrate advanced algorithms, such as Graph Neural Networks and temporal network analysis [<xref ref-type="bibr" rid="ref48">48</xref>,<xref ref-type="bibr" rid="ref49">49</xref>], to unlock new opportunities for mining deep comorbidity associations and optimizing identification performance. Sixth, certain limitations remain in our data processing. Continuous variables were neither standardized nor normalized prior to KNN imputation and model training. Although our optimal model, XGBoost, is invariant to feature scaling, the lack of normalization may have impaired the performance of distance-based algorithms, such as SVM and NN. Additionally, despite a low missing data rate (0.13%&#x2010;1.95%), applying KNN imputation without prior scaling is a methodological limitation that may introduce minor computational biases. Future studies using distance-sensitive algorithms should incorporate a comprehensive data scaling pipeline. Finally, total hospitalization costs were adjusted for economic fluctuations using the Consumer Price Index. As the general Consumer Price Index primarily tracks a standard basket of consumer goods and medical inflation often outpaces general inflation, this approach may slightly misestimate the true medical cost inflation.</p></sec><sec id="s4-3"><title>Conclusions</title><p>By integrating comorbidity network analysis with machine learning, this research used hospital discharge data to develop a high-performance framework for the identification of high-cost stroke patients. Our results confirm that, compared to stroke type within baseline features and conventional comorbidity features, network topological features are able to capture deeper information regarding disease complexity. The inclusion of network features improved model performance, with the XGBoost model demonstrating the best identification performance. Furthermore, SHAP analysis identified strong associations between high costs and relevant features while also revealing complex nonlinear interactions among these variables. These findings indicate that this framework demonstrates promising application prospects for cost-risk stratification in patients with stroke. If validated externally and prospectively in the future, this model could provide valuable decision support for exploring early identification of medical costs and stratified intervention strategies, thereby assisting in the optimization of medical resource allocation.</p></sec></sec></body><back><ack><p>The authors would like to thank all the staff who contributed to this study.</p></ack><notes><sec><title>Funding</title><p>No external financial support or grants were received from any public, commercial, or not-for-profit entities for the research, authorship, or publication of this article.</p></sec><sec><title>Data Availability</title><p>The datasets generated or analyzed during this study are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Acquisition of data: YY, PAY</p><p>Concept and design: YY, RW, PAY</p><p>Data collection and cleaning: HS, MZ</p><p>Review and editing: YY</p><p>Statistical analysis and data visualization: HS, MZ, JX</p><p>Writing original draft: HS, JX</p><p>Yilong Yang and Haohui Shen contributed equally to this work and should be regarded as joint first authors.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the curve</p></def></def-item><def-item><term id="abb2">CCI</term><def><p>Charlson Comorbidity Index</p></def></def-item><def-item><term id="abb3">DDCN</term><def><p>distance-based disease-cost network</p></def></def-item><def-item><term id="abb4">DT</term><def><p>Decision Tree</p></def></def-item><def-item><term id="abb5">DVA</term><def><p>decision curve analysis</p></def></def-item><def-item><term id="abb6">ECI</term><def><p>Elixhauser Comorbidity Index</p></def></def-item><def-item><term id="abb7"><italic>ICD-10</italic></term><def><p><italic>International Classification of Diseases, Tenth Revision</italic></p></def></def-item><def-item><term id="abb8">KNN</term><def><p>K-Nearest Neighbors</p></def></def-item><def-item><term id="abb9">LOS</term><def><p>length of stay</p></def></def-item><def-item><term id="abb10">NHCP</term><def><p>normalized high-cost propensity</p></def></def-item><def-item><term id="abb11">NN</term><def><p>Neural Network</p></def></def-item><def-item><term id="abb12">OR</term><def><p>odds ratio</p></def></def-item><def-item><term id="abb13">PCN</term><def><p>phenotypic comorbidity network</p></def></def-item><def-item><term id="abb14">RF</term><def><p>Random Forest</p></def></def-item><def-item><term id="abb15">SHAP</term><def><p>Shapley Additive Explanations</p></def></def-item><def-item><term id="abb16">SMOTE</term><def><p>Synthetic Minority Over-sampling Technique</p></def></def-item><def-item><term id="abb17">SVM</term><def><p>Support Vector Machine</p></def></def-item><def-item><term id="abb18">XGBoost</term><def><p>Extreme Gradient Boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feigin</surname><given-names>VL</given-names> </name><name name-style="western"><surname>Abate</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Abate</surname><given-names>YH</given-names> </name><etal/></person-group><article-title>Global, regional, and national burden of stroke and its risk factors, 1990&#x2013;2021: a systematic analysis for the Global Burden of Disease Study 2021</article-title><source>Lancet Neurol</source><year>2024</year><month>10</month><volume>23</volume><issue>10</issue><fpage>973</fpage><lpage>1003</lpage><pub-id pub-id-type="doi">10.1016/S1474-4422(24)00369-7</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feigin</surname><given-names>VL</given-names> </name><name name-style="western"><surname>Brainin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Norrving</surname><given-names>B</given-names> </name><etal/></person-group><article-title>World Stroke Organization: Global Stroke Fact Sheet 2025</article-title><source>Int J Stroke</source><year>2025</year><month>02</month><volume>20</volume><issue>2</issue><fpage>132</fpage><lpage>144</lpage><pub-id pub-id-type="doi">10.1177/17474930241308142</pub-id><pub-id pub-id-type="medline">39635884</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>J</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chai</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Estimating the economic burden of stroke in China: a cost-of-illness study</article-title><source>BMJ Open</source><year>2024</year><month>03</month><day>13</day><volume>14</volume><issue>3</issue><fpage>e080634</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2023-080634</pub-id><pub-id pub-id-type="medline">38485178</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garfinkel</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Riley</surname><given-names>GF</given-names> </name><name name-style="western"><surname>Iannacchione</surname><given-names>VG</given-names> </name></person-group><article-title>High-cost users of medical care</article-title><source>Health Care Financ Rev</source><year>1988</year><volume>9</volume><issue>4</issue><fpage>41</fpage><lpage>52</lpage><pub-id pub-id-type="medline">10312631</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tanke</surname><given-names>MAC</given-names> </name><name name-style="western"><surname>Feyman</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Bernal-Delgado</surname><given-names>E</given-names> </name><etal/></person-group><article-title>A challenge to all. A primer on inter-country differences of high-need, high-cost patients</article-title><source>PLOS ONE</source><year>2019</year><volume>14</volume><issue>6</issue><fpage>e0217353</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0217353</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Punjabi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Marszalek</surname><given-names>K</given-names> </name><name name-style="western"><surname>Beaney</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Categorising high-cost high-need children and young people</article-title><source>Arch Dis Child</source><year>2022</year><month>04</month><volume>107</volume><issue>4</issue><fpage>346</fpage><lpage>350</lpage><pub-id pub-id-type="doi">10.1136/archdischild-2021-321654</pub-id><pub-id pub-id-type="medline">34535444</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>X</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>M</given-names> </name></person-group><article-title>Concentration and persistence of healthcare spending: evidence from China</article-title><source>Sustainability</source><year>2021</year><volume>13</volume><issue>11</issue><fpage>5761</fpage><pub-id pub-id-type="doi">10.3390/su13115761</pub-id><pub-id pub-id-type="medline">36778665</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McWilliams</surname><given-names>JM</given-names> </name><name name-style="western"><surname>Schwartz</surname><given-names>AL</given-names> </name></person-group><article-title>Focusing on high-cost patients - the key to addressing high costs?</article-title><source>N Engl J Med</source><year>2017</year><month>03</month><day>2</day><volume>376</volume><issue>9</issue><fpage>807</fpage><lpage>809</lpage><pub-id pub-id-type="doi">10.1056/NEJMp1612779</pub-id><pub-id pub-id-type="medline">28249127</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lee</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Muratov</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tarride</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Holbrook</surname><given-names>AM</given-names> </name></person-group><article-title>Managing high-cost healthcare users: the international search for effective evidence-supported strategies</article-title><source>J Am Geriatr Soc</source><year>2018</year><month>05</month><volume>66</volume><issue>5</issue><fpage>1002</fpage><lpage>1008</lpage><pub-id pub-id-type="doi">10.1111/jgs.15257</pub-id><pub-id pub-id-type="medline">29427509</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bailey</surname><given-names>JE</given-names> </name><name name-style="western"><surname>Surbhi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wan</surname><given-names>JY</given-names> </name><etal/></person-group><article-title>Effect of intensive interdisciplinary transitional care for high-need, high-cost patients on quality, outcomes, and costs: a quasi-experimental study</article-title><source>J Gen Intern Med</source><year>2019</year><month>09</month><volume>34</volume><issue>9</issue><fpage>1815</fpage><lpage>1824</lpage><pub-id pub-id-type="doi">10.1007/s11606-019-05082-8</pub-id><pub-id pub-id-type="medline">31270786</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Quinton</surname><given-names>JK</given-names> </name><name name-style="western"><surname>Jackson</surname><given-names>N</given-names> </name><name name-style="western"><surname>Mangione</surname><given-names>CM</given-names> </name><etal/></person-group><article-title>Differential impact of a plan-led standardized complex care management intervention on subgroups of high-cost high-need Medicaid patients</article-title><source>Popul Health Manag</source><year>2023</year><month>04</month><volume>26</volume><issue>2</issue><fpage>100</fpage><lpage>106</lpage><pub-id pub-id-type="doi">10.1089/pop.2022.0271</pub-id><pub-id pub-id-type="medline">37071688</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Blumenthal</surname><given-names>D</given-names> </name><name name-style="western"><surname>Chernof</surname><given-names>B</given-names> </name><name name-style="western"><surname>Fulmer</surname><given-names>T</given-names> </name><name name-style="western"><surname>Lumpkin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Selberg</surname><given-names>J</given-names> </name></person-group><article-title>Caring for high-need, high-cost patients - an urgent priority</article-title><source>N Engl J Med</source><year>2016</year><month>09</month><day>8</day><volume>375</volume><issue>10</issue><fpage>909</fpage><lpage>911</lpage><pub-id pub-id-type="doi">10.1056/NEJMp1608511</pub-id><pub-id pub-id-type="medline">27602661</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chang</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>R</given-names> </name><name name-style="western"><surname>Berkman</surname><given-names>ND</given-names> </name></person-group><article-title>Unpacking complex interventions that manage care for high-need, high-cost patients: a realist review</article-title><source>BMJ Open</source><year>2022</year><month>06</month><day>9</day><volume>12</volume><issue>6</issue><fpage>e058539</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2021-058539</pub-id><pub-id pub-id-type="medline">35680272</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wammes</surname><given-names>JJG</given-names> </name><name name-style="western"><surname>van der Wees</surname><given-names>PJ</given-names> </name><name name-style="western"><surname>Tanke</surname><given-names>MAC</given-names> </name><name name-style="western"><surname>Westert</surname><given-names>GP</given-names> </name><name name-style="western"><surname>Jeurissen</surname><given-names>PPT</given-names> </name></person-group><article-title>Systematic review of high-cost patients&#x2019; characteristics and healthcare utilisation</article-title><source>BMJ Open</source><year>2018</year><month>09</month><day>8</day><volume>8</volume><issue>9</issue><fpage>e023113</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2018-023113</pub-id><pub-id pub-id-type="medline">30196269</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tu</surname><given-names>X</given-names> </name><name name-style="western"><surname>Luo</surname><given-names>X</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name></person-group><article-title>Machine-learning-based cost prediction models for inpatients with mental disorders in China</article-title><source>BMC Psychiatry</source><year>2025</year><month>01</month><day>9</day><volume>25</volume><issue>1</issue><fpage>33</fpage><pub-id pub-id-type="doi">10.1186/s12888-024-06358-y</pub-id><pub-id pub-id-type="medline">39789477</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Li</surname><given-names>L</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>J</given-names> </name><name name-style="western"><surname>Sanderson</surname><given-names>M</given-names> </name></person-group><article-title>Identifying and understanding determinants of high healthcare costs for breast cancer: a quantile regression machine learning approach</article-title><source>BMC Health Serv Res</source><year>2020</year><month>11</month><day>23</day><volume>20</volume><issue>1</issue><fpage>1066</fpage><pub-id pub-id-type="doi">10.1186/s12913-020-05936-6</pub-id><pub-id pub-id-type="medline">33228683</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Osawa</surname><given-names>I</given-names> </name><name name-style="western"><surname>Goto</surname><given-names>T</given-names> </name><name name-style="western"><surname>Yamamoto</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tsugawa</surname><given-names>Y</given-names> </name></person-group><article-title>Machine-learning-based prediction models for high-need high-cost patients using nationwide clinical and claims data</article-title><source>NPJ Digit Med</source><year>2020</year><month>11</month><day>11</day><volume>3</volume><issue>1</issue><fpage>148</fpage><pub-id pub-id-type="doi">10.1038/s41746-020-00354-8</pub-id><pub-id pub-id-type="medline">33299137</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Long</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Changes in the prevalence of hospitalization and comorbidity in US adults with stroke: a three decade cross-sectional and birth cohort analysis</article-title><source>Int J Stroke</source><year>2016</year><month>12</month><volume>11</volume><issue>9</issue><fpage>987</fpage><lpage>998</lpage><pub-id pub-id-type="doi">10.1177/1747493016660107</pub-id><pub-id pub-id-type="medline">27412189</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>E</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>TH</given-names> </name><name name-style="western"><surname>Koo</surname><given-names>H</given-names> </name><name name-style="western"><surname>Yoo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Heo</surname><given-names>JH</given-names> </name><name name-style="western"><surname>Nam</surname><given-names>HS</given-names> </name></person-group><article-title>Heterogeneity in costs and prognosis for acute ischemic stroke treatment by comorbidities</article-title><source>J Neurol</source><year>2019</year><month>06</month><volume>266</volume><issue>6</issue><fpage>1429</fpage><lpage>1438</lpage><pub-id pub-id-type="doi">10.1007/s00415-019-09278-0</pub-id><pub-id pub-id-type="medline">30879136</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Corraini</surname><given-names>P</given-names> </name><name name-style="western"><surname>Sz&#x00E9;pligeti</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Henderson</surname><given-names>VW</given-names> </name><name name-style="western"><surname>Ording</surname><given-names>AG</given-names> </name><name name-style="western"><surname>Horv&#x00E1;th-Puh&#x00F3;</surname><given-names>E</given-names> </name><name name-style="western"><surname>S&#x00F8;rensen</surname><given-names>HT</given-names> </name></person-group><article-title>Comorbidity and the increased mortality after hospitalization for stroke: a population-based cohort study</article-title><source>J Thromb Haemost</source><year>2018</year><month>02</month><volume>16</volume><issue>2</issue><fpage>242</fpage><lpage>252</lpage><pub-id pub-id-type="doi">10.1111/jth.13908</pub-id><pub-id pub-id-type="medline">29171148</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hwang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Chow</surname><given-names>A</given-names> </name><name name-style="western"><surname>Lye</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Wong</surname><given-names>CS</given-names> </name></person-group><article-title>Administrative data is as good as medical chart review for comorbidity ascertainment in patients with infections in Singapore</article-title><source>Epidemiol Infect</source><year>2016</year><month>07</month><volume>144</volume><issue>9</issue><fpage>1999</fpage><lpage>2005</lpage><pub-id pub-id-type="doi">10.1017/S0950268815003271</pub-id><pub-id pub-id-type="medline">26758244</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tang</surname><given-names>PL</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>HS</given-names> </name><name name-style="western"><surname>Hsu</surname><given-names>CJ</given-names> </name></person-group><article-title>Predicting in-hospital mortality for dementia patients after hip fracture surgery - A comparison between the Charlson Comorbidity Index (CCI) and the Elixhauser Comorbidity Index</article-title><source>J Orthop Sci</source><year>2021</year><month>05</month><volume>26</volume><issue>3</issue><fpage>396</fpage><lpage>402</lpage><pub-id pub-id-type="doi">10.1016/j.jos.2020.04.005</pub-id><pub-id pub-id-type="medline">32482586</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>J</given-names> </name><name name-style="western"><surname>Yue</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>G</given-names> </name></person-group><article-title>Interpretable machine learning based on the Charlson comorbidity index predicts 28-day mortality in acute hypercapnic respiratory failure</article-title><source>Sci Rep</source><year>2025</year><volume>16</volume><issue>1</issue><fpage>3335</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-33251-9</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Charlson</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Pompei</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ales</surname><given-names>KL</given-names> </name><name name-style="western"><surname>MacKenzie</surname><given-names>CR</given-names> </name></person-group><article-title>A new method of classifying prognostic comorbidity in longitudinal studies: development and validation</article-title><source>J Chronic Dis</source><year>1987</year><volume>40</volume><issue>5</issue><fpage>373</fpage><lpage>383</lpage><pub-id pub-id-type="doi">10.1016/0021-9681(87)90171-8</pub-id><pub-id pub-id-type="medline">3558716</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Elixhauser</surname><given-names>A</given-names> </name><name name-style="western"><surname>Steiner</surname><given-names>C</given-names> </name><name name-style="western"><surname>Harris</surname><given-names>DR</given-names> </name><name name-style="western"><surname>Coffey</surname><given-names>RM</given-names> </name></person-group><article-title>Comorbidity measures for use with administrative data</article-title><source>Med Care</source><year>1998</year><month>01</month><volume>36</volume><issue>1</issue><fpage>8</fpage><lpage>27</lpage><pub-id pub-id-type="doi">10.1097/00005650-199801000-00004</pub-id><pub-id pub-id-type="medline">9431328</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baneshi</surname><given-names>MR</given-names> </name><name name-style="western"><surname>Dobson</surname><given-names>A</given-names> </name><name name-style="western"><surname>Mishra</surname><given-names>GD</given-names> </name></person-group><article-title>Choices of measures of association affect the visualisation and composition of the multimorbidity networks</article-title><source>BMC Med Res Methodol</source><year>2024</year><month>07</month><day>23</day><volume>24</volume><issue>1</issue><fpage>157</fpage><pub-id pub-id-type="doi">10.1186/s12874-024-02286-3</pub-id><pub-id pub-id-type="medline">39044152</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Garc&#x00ED;a Del Valle</surname><given-names>EP</given-names> </name><name name-style="western"><surname>Lagunes Garc&#x00ED;a</surname><given-names>G</given-names> </name><name name-style="western"><surname>Prieto Santamar&#x00ED;a</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zanin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Menasalvas Ruiz</surname><given-names>E</given-names> </name><name name-style="western"><surname>Rodr&#x00ED;guez-Gonz&#x00E1;lez</surname><given-names>A</given-names> </name></person-group><article-title>Disease networks and their contribution to disease understanding: a review of their evolution, techniques and data sources</article-title><source>J Biomed Inform</source><year>2019</year><month>06</month><volume>94</volume><fpage>103206</fpage><pub-id pub-id-type="doi">10.1016/j.jbi.2019.103206</pub-id><pub-id pub-id-type="medline">31077818</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jones</surname><given-names>I</given-names> </name><name name-style="western"><surname>Cocker</surname><given-names>F</given-names> </name><name name-style="western"><surname>Jose</surname><given-names>M</given-names> </name><name name-style="western"><surname>Charleston</surname><given-names>M</given-names> </name><name name-style="western"><surname>Neil</surname><given-names>AL</given-names> </name></person-group><article-title>Methods of analysing patterns of multimorbidity using network analysis: a scoping review</article-title><source>J Public Health (Berl)</source><year>2023</year><month>08</month><volume>31</volume><issue>8</issue><fpage>1217</fpage><lpage>1223</lpage><pub-id pub-id-type="doi">10.1007/s10389-021-01685-w</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Yip</surname><given-names>PSF</given-names> </name></person-group><article-title>Predicting post-discharge self-harm incidents using disease comorbidity networks: a retrospective machine learning study</article-title><source>J Affect Disord</source><year>2020</year><month>12</month><day>1</day><volume>277</volume><fpage>402</fpage><lpage>409</lpage><pub-id pub-id-type="doi">10.1016/j.jad.2020.08.044</pub-id><pub-id pub-id-type="medline">32866798</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>M</given-names> </name></person-group><article-title>Network analytics and machine learning for predicting length of stay in elderly patients with chronic diseases at point of admission</article-title><source>BMC Med Inform Decis Mak</source><year>2022</year><month>03</month><day>10</day><volume>22</volume><issue>1</issue><fpage>62</fpage><pub-id pub-id-type="doi">10.1186/s12911-022-01802-z</pub-id><pub-id pub-id-type="medline">35272654</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Qiu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>L</given-names> </name></person-group><article-title>Early prediction of high-cost inpatients with ischemic heart disease using network analytics and machine learning</article-title><source>Expert Syst Appl</source><year>2022</year><month>12</month><volume>210</volume><fpage>118541</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2022.118541</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>He</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Assessment of stress hyperglycemia ratio to predict all-cause mortality in patients with critical cerebrovascular disease: a retrospective cohort study from the MIMIC-IV database</article-title><source>Cardiovasc Diabetol</source><year>2025</year><volume>24</volume><issue>1</issue><fpage>58</fpage><pub-id pub-id-type="doi">10.1186/s12933-025-02613-y</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Dharmarajan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hsieh</surname><given-names>AF</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Diagnoses and timing of 30-day readmissions after hospitalization for heart failure, acute myocardial infarction, or pneumonia</article-title><source>JAMA</source><year>2013</year><month>01</month><day>23</day><volume>309</volume><issue>4</issue><fpage>355</fpage><lpage>363</lpage><pub-id pub-id-type="doi">10.1001/jama.2012.216476</pub-id><pub-id pub-id-type="medline">23340637</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name></person-group><article-title>Comparison of 30-day planned and unplanned readmissions in a tertiary teaching hospital in China</article-title><source>BMC Health Serv Res</source><year>2023</year><month>03</month><day>6</day><volume>23</volume><issue>1</issue><fpage>213</fpage><pub-id pub-id-type="doi">10.1186/s12913-023-09193-1</pub-id><pub-id pub-id-type="medline">36879245</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Srinivasan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Currim</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ram</surname><given-names>S</given-names> </name></person-group><article-title>Predicting high-cost patients at point of admission using network science</article-title><source>IEEE J Biomed Health Inform</source><year>2018</year><month>11</month><volume>22</volume><issue>6</issue><fpage>1970</fpage><lpage>1977</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2017.2783049</pub-id><pub-id pub-id-type="medline">29990022</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Iyengar</surname><given-names>V</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Predicting future high-cost schizophrenia patients using high-dimensional administrative data</article-title><source>Front Psychiatry</source><year>2017</year><volume>8</volume><fpage>114</fpage><pub-id pub-id-type="doi">10.3389/fpsyt.2017.00114</pub-id><pub-id pub-id-type="medline">28713293</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fleishman</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Cohen</surname><given-names>JW</given-names> </name></person-group><article-title>Using information on clinical conditions to predict high-cost patients</article-title><source>Health Serv Res</source><year>2010</year><month>04</month><volume>45</volume><issue>2</issue><fpage>532</fpage><lpage>552</lpage><pub-id pub-id-type="doi">10.1111/j.1475-6773.2009.01080.x</pub-id><pub-id pub-id-type="medline">20132341</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salmi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Atif</surname><given-names>D</given-names> </name><name name-style="western"><surname>Oliva</surname><given-names>D</given-names> </name><name name-style="western"><surname>Abraham</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ventura</surname><given-names>S</given-names> </name></person-group><article-title>Handling imbalanced medical datasets: review of a decade of research</article-title><source>Artif Intell Rev</source><year>2024</year><volume>57</volume><issue>10</issue><pub-id pub-id-type="doi">10.1007/s10462-024-10884-2</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Identification and validation of an explainable early-stage chronic kidney disease prediction model: a multicenter retrospective study</article-title><source>EClinicalMedicine</source><year>2025</year><month>06</month><volume>84</volume><fpage>103286</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2025.103286</pub-id><pub-id pub-id-type="medline">40567347</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Identification and validation of an explainable prediction model of acute kidney injury with prognostic implications in critically ill children: a prospective multicenter cohort study</article-title><source>EClinicalMedicine</source><year>2024</year><month>02</month><volume>68</volume><fpage>102409</fpage><pub-id pub-id-type="doi">10.1016/j.eclinm.2023.102409</pub-id><pub-id pub-id-type="medline">38273888</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Relationship between atherogenic index of plasma and length of stay in critically ill patients with atherosclerotic cardiovascular disease: a retrospective cohort study and predictive modeling based on machine learning</article-title><source>Cardiovasc Diabetol</source><year>2025</year><month>02</month><day>28</day><volume>24</volume><issue>1</issue><fpage>95</fpage><pub-id pub-id-type="doi">10.1186/s12933-025-02654-3</pub-id><pub-id pub-id-type="medline">40022165</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>X</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>Predicting high-need high-cost pediatric hospitalized patients in China based on machine learning methods</article-title><source>Sci Rep</source><year>2025</year><volume>15</volume><issue>1</issue><fpage>16006</fpage><pub-id pub-id-type="doi">10.1038/s41598-025-99546-z</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Langenberger</surname><given-names>B</given-names> </name><name name-style="western"><surname>Schulte</surname><given-names>T</given-names> </name><name name-style="western"><surname>Groene</surname><given-names>O</given-names> </name></person-group><article-title>The application of machine learning to predict high-cost patients: a performance-comparison of different models using healthcare claims data</article-title><source>PLoS ONE</source><year>2023</year><volume>18</volume><issue>1</issue><fpage>e0279540</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0279540</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nghiem</surname><given-names>N</given-names> </name><name name-style="western"><surname>Atkinson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>BP</given-names> </name><name name-style="western"><surname>Tran-Duy</surname><given-names>A</given-names> </name><name name-style="western"><surname>Wilson</surname><given-names>N</given-names> </name></person-group><article-title>Predicting high health-cost users among people with cardiovascular disease using machine learning and nationwide linked social administrative datasets</article-title><source>Health Econ Rev</source><year>2023</year><month>02</month><day>4</day><volume>13</volume><issue>1</issue><fpage>9</fpage><pub-id pub-id-type="doi">10.1186/s13561-023-00422-1</pub-id><pub-id pub-id-type="medline">36738348</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khan</surname><given-names>SU</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>MZ</given-names> </name><name name-style="western"><surname>Khan</surname><given-names>MU</given-names> </name><etal/></person-group><article-title>Clinical and economic burden of stroke among young, midlife, and older adults in the United States, 2002-2017</article-title><source>Mayo Clin Proc Innov Qual Outcomes</source><year>2021</year><month>04</month><volume>5</volume><issue>2</issue><fpage>431</fpage><lpage>441</lpage><pub-id pub-id-type="doi">10.1016/j.mayocpiqo.2021.01.015</pub-id><pub-id pub-id-type="medline">33997639</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Deng</surname><given-names>G</given-names> </name><name name-style="western"><surname>Meng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>H</given-names> </name></person-group><article-title>Hospitalization expenditures and out-of-pocket expenses in patients with stroke in Northeast China, 2015-2017: a pooled cross-sectional study</article-title><source>Front Pharmacol</source><year>2020</year><volume>11</volume><fpage>596183</fpage><pub-id pub-id-type="doi">10.3389/fphar.2020.596183</pub-id><pub-id pub-id-type="medline">33613278</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Man</surname><given-names>X</given-names> </name><name name-style="western"><surname>Nicholas</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Utilisation of health services among urban patients who had an ischaemic stroke with different health insurance - a cross-sectional study in China</article-title><source>BMJ Open</source><year>2020</year><month>10</month><volume>10</volume><issue>10</issue><fpage>e040437</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2020-040437</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Hoyt</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chatterjee</surname><given-names>N</given-names> </name><name name-style="western"><surname>Battaglia</surname><given-names>F</given-names> </name><name name-style="western"><surname>Basu</surname><given-names>P</given-names> </name></person-group><article-title>Medical applications of graph convolutional networks using electronic health records: a survey</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 13, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2502.09781</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Gardinazzi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>March</surname><given-names>RG</given-names> </name><name name-style="western"><surname>Kalahasti</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ramirez</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Neri</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nguyen</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Characterization of diseases in temporal comorbidity networks</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 29, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2506.22136</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary data regarding the configuration and evaluation of the machine learning models, specifically containing the hyperparameter search spaces and the resulting optimal parameters for the machine learning models; a forest plot illustrating the classification performance for the models across different stroke types.</p><media xlink:href="medinform_v14i1e93680_app1.docx" xlink:title="DOCX File, 84 KB"/></supplementary-material></app-group></back></article>