<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e83544</article-id><article-id pub-id-type="doi">10.2196/83544</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Reinforcement Learning&#x2013;Based Temporal Knowledge Graph Reasoning for Predicting Chronic Gastritis Diagnosis and Treatment: Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Qu</surname><given-names>Xiaolong</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sun</surname><given-names>Zhou</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wang</surname><given-names>Yuhang</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Haiyu</given-names></name><degrees>MS</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yao</surname><given-names>Lei</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Li</surname><given-names>Dongmei</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Song</surname><given-names>Guanli</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Runshun</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Xiaoping</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib></contrib-group><aff id="aff1"><institution>School of Information Science and Technology, Beijing Forestry University</institution><addr-line>35 Qinghua East Road, Haidian District</addr-line><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff2"><institution>Hebei Key Laboratory of Smart National Park</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff3"><institution>National Data Center of Tradition Chinese Medicine of China, Academy of Chinese Medical Sciences</institution><addr-line>Beijing</addr-line><country>China</country></aff><aff id="aff4"><institution>University of Wisconsin&#x2013;Milwaukee</institution><addr-line>Milwaukee</addr-line><addr-line>WI</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Benis</surname><given-names>Arriel</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Liu</surname><given-names>Kangzheng</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Chen</surname><given-names>Kuan-Fu</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Lim</surname><given-names>Min Hyuk</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Dongmei Li, PhD, School of Information Science and Technology, Beijing Forestry University, 35 Qinghua East Road, Haidian District, Beijing, 100091, China, 86 13120253485; <email>lidongmei@bjfu.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>11</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e83544</elocation-id><history><date date-type="received"><day>07</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>13</day><month>08</month><year>2026</year></date><date date-type="accepted"><day>13</day><month>08</month><year>2026</year></date></history><copyright-statement>&#x00A9; Xiaolong Qu, Zhou Sun, Yuhang Wang, Haiyu Liu, Lei Yao, Dongmei Li, Guanli Song, Runshun Zhang, Xiaoping Zhang. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 11.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e83544"/><abstract><sec><title>Background</title><p>The clinical progression of chronic gastritis involves intricate temporal dependencies, which makes it difficult to capture both the dynamic trajectory of the disease and the underlying relationships among medical events using conventional methods.</p></sec><sec><title>Objective</title><p>This study aims to dynamically predict chronic gastritis diagnosis. We propose RL4TKGR, a reinforcement learning&#x2013;based temporal knowledge graph (TKG) reasoning model, to predict a chronic gastritis diagnosis.</p></sec><sec sec-type="methods"><title>Methods</title><p>RL4TKGR incorporates a reinforcement learning policy network that uses a dual-path encoding module to model historical and nonhistorical diagnostic information separately. RL4TKGR further integrates a dual-channel reward function with a dynamic weight allocation mechanism, which adaptively balances the two information sources. This design addresses the strategic bias problem and enables interpretable reasoning. The chronic gastritis temporal knowledge graph (CG-TKG) was constructed from chronic gastritis diagnosis and treatment records of 17,906 patients comprising 38,360 visits from March 2009 to October 2022.</p></sec><sec sec-type="results"><title>Results</title><p>Experiments on the self-constructed CG-TKG resulted in RL4TKGR achieving the highest mean reciprocal rank (MRR) on CG-TKG-4 (51.66, SD 0.76), CG-TKG-8 (53.18, SD 0.79), and CG-TKG-12 (56.95, SD 0.84). After adding recent TKG reasoning baselines, adaptive path-memory network (DaeMon) achieved the strongest baseline MRR for all 3 subsets, with MRR values of 43.26 (SD 0.95), 51.76 (SD 1.12), and 54.87 (SD 1.17), respectively. Compared with DaeMon, the overall paired <italic>t</italic> tests across MRR and Hits@1/3/10 yielded &#x1D443;&#x003C;.001 for all 3 subsets. Ablation experiments and case analyses further supported the contribution of the historical modeling components and the practical utility of the model for predicting disease subtypes and therapeutic medications.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>This study improves chronic gastritis diagnosis and treatment prediction by integrating a history-aware dual-path policy network with a dynamic event balance factor in reinforcement learning&#x2013;based TKG reasoning.</p></sec></abstract><kwd-group><kwd>temporal knowledge graph reasoning</kwd><kwd>reinforcement learning</kwd><kwd>chronic gastritis</kwd><kwd>diagnosis prediction</kwd><kwd>traditional Chinese medicine</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Chronic gastritis is a chronic inflammation of the gastric mucosa caused by multiple pathogenic factors. Its spectrum of precancerous lesions, including gastric atrophy, intestinal metaplasia, and dysplasia, has conferred independent carcinogenic risks [<xref ref-type="bibr" rid="ref1">1</xref>]. Consequently, elucidating the evolutionary trajectory of chronic gastritis and accurately predicting its progression are of critical importance for the early detection and intervention of gastric cancer, which underscores the significant clinical value of this task.</p><p>In recent years, artificial intelligence (AI) technologies have been extensively explored for the diagnosis and treatment of chronic gastritis. In prediction studies based on static features, several models were used to analyze gastroscopic images and demonstrated high accuracy and efficiency in the diagnosis of gastric diseases such as chronic atrophic gastritis, which significantly improved the accuracy of real-time diagnosis [<xref ref-type="bibr" rid="ref2">2</xref>-<xref ref-type="bibr" rid="ref4">4</xref>]. In the study by Tao et al [<xref ref-type="bibr" rid="ref5">5</xref>], the model not only was able to diagnose gastric mucosal atrophy but also performed risk stratification based on the extent of atrophy; this provided decision support for subsequent monitoring intervals. However, static prediction models struggle to capture the dynamic evolution trends of diseases and fail to adequately consider the temporal characteristics inherent in clinical progression. Therefore, current research has gradually shifted from static, point-in-time diagnosis toward dynamic risk prediction and disease course evolution analysis based on temporal medical data. Existing studies leveraged electronic medical record (EMR) data and applied temporal analysis techniques to predict the progression risk from atrophic gastritis to gastric cancer while also quantifying relevant disease trajectories and high-risk factors [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. Nevertheless, temporal medical data primarily record sequences of indicator values changing over time and inherently lack the capacity to organize information into structured tuples. This limitation constrains the ability to model deeper associations among medical events.</p><p>Temporal knowledge graphs (TKGs) extend the static knowledge graph framework by introducing the temporal dimension [<xref ref-type="bibr" rid="ref8">8</xref>], and they represent the evolutionary processes of dynamic events using quadruples of the form <inline-formula><mml:math id="ieqn1"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>o</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. In medical scenarios, knowledge graphs can provide physicians with reliable auxiliary diagnostic tools and offer good explainability [<xref ref-type="bibr" rid="ref9">9</xref>]. Moreover, TKGs can model the dynamic evolution of relationships between entities and demonstrate key value for disease progression prediction and drug interaction tracking [<xref ref-type="bibr" rid="ref10">10</xref>]. Static pathological features are limited in capturing the dynamic trends of disease progression and overlook the clinical evolution of the disease course, whereas TKGs can embed time stamps into entity states to capture the progressive nature of chronic gastritis. In contrast to temporal data, which capture variations in indicator values over time but cannot organize information into structured tuples, TKGs enable the representation of the clinical evolution of chronic gastritis, thereby improving both the accuracy and interpretability of prediction. Nevertheless, studies that leverage TKGs to predict the dynamic progression of chronic gastritis remain scarce.</p><p>In clinical practice, historical diagnostic information for patients plays an essential role in predicting diagnoses for subsequent patients with the same disease. The patterns of disease representation similarity in patient symptoms [<xref ref-type="bibr" rid="ref11">11</xref>] provide a theoretical foundation for temporal diagnosis and treatment modeling. Reinforcement learning excels at deriving optimal strategies through interaction with the environment, which makes it well suited to simulate the clinical decision-making process in which physicians formulate future decisions based on the historical medical records and current conditions of patients. Furthermore, the medical decision-making process can be regarded as a sequential decision-making problem of finding the optimal path within massive information, which aligns closely with the modeling paradigm of reinforcement learning. Therefore, we constructed a chronic gastritis temporal knowledge graph (CG-TKG) based on real-world dynamic diagnostic and treatment data. By modeling and analyzing the historical medical records of patients and incorporating the characteristics of clinical data, we proposed a reinforcement learning&#x2013;based TKG reasoning model named RL4TKGR. The main methodological novelty of RL4TKGR lies in its architecture-specific separation of historical and nonhistorical diagnostic paths and its dynamic &#x03B2;-based weighting mechanism, rather than in the general use of TKG reasoning or reinforcement learning alone. This model is designed to improve prediction of disease subtypes and therapeutic drugs for chronic gastritis on the evaluated CG-TKG task.</p><p>To summarize, the contributions of this paper are as follows: (1) We proposed RL4TKGR for chronic gastritis diagnosis prediction. We designed a reinforcement learning policy network that exploits associations within historical diagnostic data to predict chronic gastritis diagnosis. RL4TKGR captures both the features from the historical medical records of patients and the latent features of nonhistorical diagnoses, which enables interpretable reasoning on the CG-TKG. (2) We proposed a dual-channel aggregated reward function and a dual-path encoded policy network based on historical diagnosis. By constructing reward functions for historical and nonhistorical information, the model guided the learning of the corresponding paths within the policy network. Furthermore, the event balance factor was used to dynamically and interpretably adjust the weight allocation in the dual-channel aggregation reward function and the action scoring module, which effectively addressed the strategy bias issue caused by static weight allocation. (3) On the self-constructed CG-TKG datasets, RL4TKGR achieved mean reciprocal rank (MRR) values of 51.66 on CG-TKG-4, 53.18 on CG-TKG-8, and 56.95 on CG-TKG-12, compared with the strongest MRR baselines of 43.26, 51.76, and 54.87, respectively, after adding recent TKG reasoning baselines. RL4TKGR also achieved Hits@10 values of 60.15, 67.26, and 71.79 on the 3 datasets, respectively. These absolute performance values indicated that RL4TKGR improves chronic gastritis diagnosis and treatment prediction on temporal clinical data.</p></sec><sec id="s1-2"><title>Related Work</title><sec id="s1-2-1"><title>TKG Reasoning</title><p>A TKG is a special type of knowledge graph that adds a temporal dimension to the traditional triple representation (entity, relation, entity), which forms a new quadruple (entity, relation, entity, time). This representation can capture the temporal evolution of relationships among entities, which is of great significance for understanding and predicting their dynamic interactions [<xref ref-type="bibr" rid="ref12">12</xref>].</p><p>The TTransE [<xref ref-type="bibr" rid="ref13">13</xref>] model extends the TransE [<xref ref-type="bibr" rid="ref14">14</xref>] model by adding a temporal dimension. Its scoring function considers temporal information, which enables the model to handle quadruples in TKGs. The TA-DistMult [<xref ref-type="bibr" rid="ref15">15</xref>] model is a temporal version of the DistMult [<xref ref-type="bibr" rid="ref16">16</xref>] model that integrates temporal information by learning a sequence encoder. This model uses a recurrent neural network (RNN) to learn time-aware representations of relation types. Similarly, TA-TransE [<xref ref-type="bibr" rid="ref15">15</xref>] is a temporal variant of TransE that incorporates temporal information into the standard embedding framework for link prediction. This model leverages an RNN to encode sequences of temporal tokens, thereby learning time-aware representations.</p><p>DE-SimplE [<xref ref-type="bibr" rid="ref17">17</xref>] equips static models with diachronic entity embedding for TKG completion. xERTE [<xref ref-type="bibr" rid="ref18">18</xref>] proposes an explainable reasoning framework for future link prediction in TKGs, which uses a temporal relation attention mechanism and a novel reverse representation update scheme. TimeTraveler (TITer) [<xref ref-type="bibr" rid="ref19">19</xref>] was the first to use path-based reinforcement learning for TKG reasoning, and it can handle unseen time stamps and entities. In addition, run-time type information (RTTI) [<xref ref-type="bibr" rid="ref20">20</xref>] proposed a new method for representing time intervals on this basis, which uses the median of 2 time stamps and embedding changes to represent time intervals.</p><p>Existing works adopt various TKG extrapolation architectures to model the evolution patterns of TKGs. Relation-enhanced graph convolutional network (RE-GCN) [<xref ref-type="bibr" rid="ref21">21</xref>] designs a recurrent evolutionary network to capture both structural dependencies and temporal patterns. To address the problem of length diversity in evolution patterns, complex evolutional network (CEN) [<xref ref-type="bibr" rid="ref22">22</xref>] combines relational graph neural networks and length-aware convolutional networks to learn evolution features of historical sequences of different lengths. Recurrent event network (RE-Net) [<xref ref-type="bibr" rid="ref23">23</xref>] models the temporal conditional probability distribution of event sequences through a recursive event encoder, while it uses a neighborhood aggregator to handle concurrent events and enable global structural reasoning. Contrastive event network (CENET) [<xref ref-type="bibr" rid="ref24">24</xref>] further considers the influence of potential unobserved factors and leverages contrastive learning to model dependencies between historical and nonhistorical events, thereby improving predictive performance.</p><p>To avoid grouping heterogeneous methods into a single category, recent TKG reasoning studies can be organized into several families. Time-aware embedding models, such as TTransE, TA-DistMult, TA-TransE, and DE-SimplE, encode temporal signals into entity or relation representations. Continuous-time models, such as Know-Evolve [<xref ref-type="bibr" rid="ref25">25</xref>] and Temporal Knowledge Graph Forecasting with Neural Ordinary Equations (TANGO) [<xref ref-type="bibr" rid="ref26">26</xref>], model event intensities or continuous dynamic embeddings rather than only discrete graph snapshots. Recurrent and graph-based extrapolation architectures, including RE-Net, RE-GCN, CEN, CENET, time-guided recurrent graph network (TiRGN) [<xref ref-type="bibr" rid="ref27">27</xref>], relation-entity twin-interact aggregation (RETIA) [<xref ref-type="bibr" rid="ref28">28</xref>], and adaptive path-memory network (DaeMon) [<xref ref-type="bibr" rid="ref29">29</xref>], emphasize future-link prediction from historical snapshots. Path-based and rule-based models, such as TITer, xERTE, and TLogic [<xref ref-type="bibr" rid="ref30">30</xref>], provide more explicit reasoning paths or temporal logical rules. RL4TKGR belongs to the history-aware path reasoning family and is tailored to discrete visit-order clinical TKGs.</p><p>Although these model families have advanced TKG reasoning, several limitations remain for visit-level clinical TKGs. Embedding and recurrent extrapolation models improve temporal prediction but often treat historical facts as global graph snapshots and provide limited patient-level path evidence. Continuous-time models can represent irregular event times, but they require reliable time-stamped event streams and are less directly aligned with deidentified visit-order clinical records. Path-based methods improve interpretability, yet existing designs generally do not explicitly separate historical diagnostic paths from current or nonhistorical clinical evidence. These gaps motivate the dual-path policy network and dynamic event balance factor in RL4TKGR.</p></sec><sec id="s1-2-2"><title>Application of TKG Reasoning in Disease Prediction</title><p>Knowledge graph reasoning technology in the medical field is evolving from static modeling to dynamic temporal analysis [<xref ref-type="bibr" rid="ref31">31</xref>]. Carvalho et al [<xref ref-type="bibr" rid="ref32">32</xref>] proposed an ontology-based framework for constructing EMR temporal graphs that achieved dynamic tracking of patient status through semantic annotation and time stamp mapping and established a foundation for the advancement of knowledge graph reasoning in the medical domain.</p><p>Considering the temporal characteristics of diabetes progression, Geng et al [<xref ref-type="bibr" rid="ref33">33</xref>] proposed an incremental long short-term memory (LSTM) that embeds entity relations through TransR, integrates graph topological structures, and achieved precise modeling of complication evolution on clinical knowledge graphs. Chaturvedi [<xref ref-type="bibr" rid="ref34">34</xref>] leveraged a pretrained transformer to extract fine-grained temporal relations from clinical texts and construct a patient-centered multisource temporal graph that provides a scalable risk prediction framework for chronically monitored diseases such as type 2 diabetes. Considering the task characteristics of long-term health risk prediction, Postiglione et al [<xref ref-type="bibr" rid="ref35">35</xref>] proposed the MedTKG framework, which integrates the dynamic temporal information from electronic health records (EHRs) with the static hierarchy of medical ontologies and validates the effectiveness of ontology-enhanced temporal reasoning on the MIMIC-III dataset. All of these efforts focused on chronic disease evolution and complication prediction. For the prediction of acute disease deterioration, Song et al [<xref ref-type="bibr" rid="ref10">10</xref>] accounted for the rapid worsening of disease trajectories by integrating temporal information into a gated recurrent unit (GRU) network while preserving graph structural properties with TransR, thereby developing a diagnostic support system capable of real-time prediction of acute diseases and their complications.</p><p>However, most medical TKG applications focus on general EHR risk prediction, complications, or acute deterioration, and few address chronic gastritis diagnosis and treatment prediction in a traditional Chinese medicine (TCM)&#x2013;specific entity space. In addition, existing clinical graph studies often emphasize predictive accuracy or ontology construction, whereas fewer studies examine how historical and current-visit evidence are balanced for individual patients.</p></sec><sec id="s1-2-3"><title>Application of Reinforcement Learning in Disease Prediction</title><p>As one of the 3 major paradigms of machine learning, reinforcement learning centers on the dynamic interaction between an agent and its environment, optimizing decision-making by maximizing target objectives through reward signals. Unlike supervised learning, reinforcement learning models the decision-making trajectory through a Markov decision process [<xref ref-type="bibr" rid="ref36">36</xref>].</p><p>The core advantage of reinforcement learning in disease prediction lies in its dynamic decision optimization capability. Compared with static prediction models, reinforcement learning can simulate the multistage process of disease development and formulate personalized intervention strategies according to the real-time status of patients. In the context of Alzheimer disease prediction, Chaudhari and Khot [<xref ref-type="bibr" rid="ref37">37</xref>] proposed CAdam-RL-DCNN, which integrates an improved Coyote optimization algorithm with the Adam optimizer and combines a reinforcement learning decision mechanism to address the problems of feature extraction efficiency and class imbalance. Zhang et al [<xref ref-type="bibr" rid="ref38">38</xref>] constructed a weighted dueling double deep Q-network, integrated clinical expert rules to guide action selection, and ensured decision reliability through doubly robust off-policy evaluation, providing a new paradigm for dynamic treatment in intensive care units.</p><p>Despite these advances, most existing TKG reasoning models in the medical domain rely on static weight fusion or fixed embedding schemes, which limits their ability to dynamically adapt to individualized diagnostic and treatment trajectories or to differentiate historical and nonhistorical clinical events, thus leading to policy bias. Likewise, current reinforcement learning&#x2013;based disease prediction models often depend on static rules or weight fusion, which constrains their capacity to dynamically capture the relevance of historical events. To address these challenges, we proposed the RL4TKGR model based on historical diagnostic information. This model effectively overcomes the bias problem of static fusion and significantly improves the accuracy and interpretability of chronic gastritis prediction.</p></sec></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Model Framework</title><p>The framework of RL4TKGR is shown in <xref ref-type="fig" rid="figure1">Figure 1</xref>, with the policy network as its core component. The policy network consists of 4 modules: diagnosis embedding module, dual-path encoding module, action scoring module, and dynamic weight allocator module. The dual-channel reward function guides the learning of the policy network. It comprises both a historical reward function and a nonhistorical reward function, which are adaptively integrated through the dynamic weight allocation module.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Framework of the proposed reinforcement learning&#x2013;based temporal knowledge graph reasoning (RL4TKGR) model. CG-TKG: chronic gastritis temporal knowledge graph; LSTM: long short-term memory.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e83544_fig01.png"/></fig></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study was approved by the Medical Ethics Committee of Guang&#x2019;anmen Hospital, China Academy of Chinese Medical Sciences (approval number: 2024&#x2010;241-KY; approval date: December 10, 2024; valid from December 10, 2024, to December 31, 2027). The ethics committee approved the study protocol and granted a waiver for informed consent. Only deidentified clinical records were used for model development and validation, and the research team has no access to direct patient identifiers. Individual-level clinical records are not publicly shared because they contain sensitive health information and are subject to institutional ethics and data governance restrictions.</p></sec><sec id="s2-3"><title>Formal Definitions</title><sec id="s2-3-1"><title>TKG</title><p>A TKG was defined as <inline-formula><mml:math id="ieqn2"><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn3"><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes a snapshot at time stamp <inline-formula><mml:math id="ieqn4"><mml:mi>t</mml:mi></mml:math></inline-formula>. Here, <inline-formula><mml:math id="ieqn5"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the entity set, <inline-formula><mml:math id="ieqn6"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the relation set, and <inline-formula><mml:math id="ieqn7"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the fact set at <inline-formula><mml:math id="ieqn8"><mml:mi>t</mml:mi></mml:math></inline-formula>. Each fact is expressed as <inline-formula><mml:math id="ieqn9"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn10"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> indicate subject and object entities, respectively, and <inline-formula><mml:math id="ieqn11"><mml:mi>r</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes their relation. In the CG-TKG, <inline-formula><mml:math id="ieqn12"><mml:mi>t</mml:mi></mml:math></inline-formula> denotes the visit number,  <inline-formula><mml:math id="ieqn13"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the set of medical entities at the <inline-formula><mml:math id="ieqn14"><mml:mi>t</mml:mi></mml:math></inline-formula>-th visit, <inline-formula><mml:math id="ieqn15"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the set of medical relations at the <inline-formula><mml:math id="ieqn16"><mml:mi>t</mml:mi></mml:math></inline-formula>-th visit, and <inline-formula><mml:math id="ieqn17"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the set of medical facts at the <italic>t</italic>-th visit.</p><p>In the CG-TKG, time stamp <italic>t</italic> is encoded as the visit-order index rather than continuous calendar time. Thus, <italic>t</italic>=4 denotes the fourth recorded clinical visit of a patient, not a specific date. This discrete encoding matches the visit-level organization of the clinical records and supports temporal snapshot construction. However, it does not directly model heterogeneous elapsed days between adjacent visits.</p></sec><sec id="s2-3-2"><title>TKG Reasoning Task</title><p>The TKG reasoning task was defined as performing link prediction in knowledge graphs at future time stamps along the temporal evolution. Given a query), the missing object entity in the query is predicted using the TKG composed of the set of known facts . In the CG-TKG reasoning task, the query) indicates that the missing medical entity is predicted based on the set of known medical facts in the CG-TKG, with a primary focus on predicting the types of chronic gastritis and corresponding therapeutic drugs. For example, the query <inline-formula><mml:math id="ieqn18"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext>patient ID</mml:mtext><mml:mo>,</mml:mo><mml:mtext>diagnosis</mml:mtext><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> indicates predicting the disease diagnosis of a patient (identified by patient ID) at the fourth clinical visit based on the set of medical facts from the first 3 visits.</p></sec></sec><sec id="s2-4"><title>Reinforcement Learning Framework</title><sec id="s2-4-1"><title>States</title><p>We defined <inline-formula><mml:math id="ieqn19"><mml:mi>S</mml:mi></mml:math></inline-formula> as the state space, where each state can be represented as a 5-tuple <inline-formula><mml:math id="ieqn20"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:mi>S</mml:mi></mml:math></inline-formula>. Here, <inline-formula><mml:math id="ieqn21"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the node accessed at step <inline-formula><mml:math id="ieqn22"><mml:mi>n</mml:mi></mml:math></inline-formula> and visible exclusively at step <inline-formula><mml:math id="ieqn23"><mml:mi>n</mml:mi></mml:math></inline-formula>, with <inline-formula><mml:math id="ieqn24"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x00A7;amp;lt;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>; <inline-formula><mml:math id="ieqn25"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> representing elements in the query <inline-formula><mml:math id="ieqn26"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>q</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mi>q</mml:mi></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mi>q</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> and globally visible. The agent uses <inline-formula><mml:math id="ieqn27"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> as the initial state.</p></sec><sec id="s2-4-2"><title>Actions</title><p>We defined <inline-formula><mml:math id="ieqn28"><mml:mi>A</mml:mi></mml:math></inline-formula> as the action space, and <inline-formula><mml:math id="ieqn29"><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as the set of optional actions at step <inline-formula><mml:math id="ieqn30"><mml:mi>n</mml:mi></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn31"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>A</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>r</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2228;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mi>r</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:msup><mml:mi>t</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msup><mml:mi>t</mml:mi><mml:mrow><mml:mo>&#x2032;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x003C;</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>. Since a patient&#x2019;s disease progresses with each visit, the agent can select actions with time stamps subsequent to <inline-formula><mml:math id="ieqn32"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> during action selection. Therefore, in the TKG reasoning task, the agent can select actions from the entire set of known facts.</p></sec><sec id="s2-4-3"><title>Transition</title><p>We defined a state transition function <inline-formula><mml:math id="ieqn33"><mml:mi>&#x03B4;</mml:mi></mml:math></inline-formula> to map from state <inline-formula><mml:math id="ieqn34"><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to <inline-formula><mml:math id="ieqn35"><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> during the state transition performed by the agent, such that <inline-formula><mml:math id="ieqn36"><mml:mi>S</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>A</mml:mi><mml:mo>&#x2192;</mml:mo><mml:mi>S</mml:mi></mml:math></inline-formula> (ie, <inline-formula><mml:math id="ieqn37"><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>.</p></sec><sec id="s2-4-4"><title>Rewards</title><p>We defined a dual-channel aggregated reward function <inline-formula><mml:math id="ieqn38"><mml:mi>R</mml:mi></mml:math></inline-formula> that integrates historical and nonhistorical diagnostic information to guide the agent toward accurate action selection.</p></sec></sec><sec id="s2-5"><title>Dual-Channel Reward Function</title><sec id="s2-5-1"><title>Historical Reward</title><p>The historical reward enables the policy network to focus on historical diagnosis information. When the predicted fact is related to historical interactions, the model incorporates these known facts into the prediction of future facts, as expressed by <xref ref-type="disp-formula" rid="E1">Equation 1</xref>:</p><disp-formula id="E1"><label>(1)</label><mml:math id="eqn1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">I</mml:mi></mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>&#x2227;</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>H</mml:mi><mml:mi>q</mml:mi></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn39"><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the set of historical entities, that is, all entities that appeared in the previous <inline-formula><mml:math id="ieqn40"><mml:mi>n</mml:mi></mml:math></inline-formula> visits. The historical occurrence frequency <inline-formula><mml:math id="ieqn41"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> of entity <inline-formula><mml:math id="ieqn42"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>H</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is calculated and participates in the reward function computation. Here, <inline-formula><mml:math id="ieqn43"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the terminal state after <inline-formula><mml:math id="ieqn44"><mml:mi>L</mml:mi></mml:math></inline-formula> reasoning steps, <inline-formula><mml:math id="ieqn45"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the terminal entity reached by the selected action trajectory, <inline-formula><mml:math id="ieqn46"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> denotes the ground-truth object entity of the query, and <inline-formula><mml:math id="ieqn47"><mml:mi>n</mml:mi></mml:math></inline-formula> denotes the number of visits before the query time stamp <inline-formula><mml:math id="ieqn48"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> that are available as patient history.</p></sec><sec id="s2-5-2"><title>Nonhistorical Reward</title><p>The nonhistorical reward does not consider the influence of historical information in the diagnosis process and allows the predicted fact to be treated as a new potential fact unrelated to historical interactions, as expressed by <xref ref-type="disp-formula" rid="E2">Equation 2</xref>:</p><disp-formula id="E2"><label>(2)</label><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>s</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mi mathvariant="normal">I</mml:mi></mml:mrow><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mi>L</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">f</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">t</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn49"><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the predicted entity of the action finally selected by the agent, and <inline-formula><mml:math id="ieqn50"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> is the correct entity in the query quadruple.</p></sec><sec id="s2-5-3"><title>Dual-Channel Aggregation Reward</title><p>The dual-channel aggregated reward function performs a weighted integration of the historical and nonhistorical reward functions, which can be expressed as <xref ref-type="disp-formula" rid="E3">Equation 3</xref>:</p><disp-formula id="E3"><label>(3)</label><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">h</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">s</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn51"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> is the aggregation weight. In RL4TKGR, <inline-formula><mml:math id="ieqn52"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> is set to the query-level value of the event balance factor <inline-formula><mml:math id="ieqn53"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> generated by the dynamic weight allocator module. In the fixed- sensitivity experiment, this adaptive value was replaced by prespecified constants to isolate the effect of the historical information weight.</p></sec><sec id="s2-5-4"><title>Policy Network</title><p>The policy network <inline-formula><mml:math id="ieqn54"><mml:msub><mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> evaluates the action <inline-formula><mml:math id="ieqn55"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> selected by the agent, where <inline-formula><mml:math id="ieqn56"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula> denotes the model parameters. This network consists of 3 modules: diagnosis embedding module, dual-path encoding module, action scoring module, and dynamic weight allocator module.</p><sec id="s2-5-4-1"><title>Diagnosis Embedding Module</title><p>We began by randomly and uniformly initializing each fact <inline-formula><mml:math id="ieqn57"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>r</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> within a low-dimensional dense vector space. Since a patient&#x2019;s disease progression is closely related to visit number, the entity and visit number were jointly represented as a diagnosis at the current step, that is, <inline-formula><mml:math id="ieqn58"><mml:msubsup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. Entities, relations, and time stamps are represented as <inline-formula><mml:math id="ieqn59"><mml:mi>e</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn60"><mml:mi>r</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn61"><mml:mi>t</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn62"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn63"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn64"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, respectively, denote their embedding dimensions.</p><p>During the patient&#x2019;s visit process, diseases evolve gradually over time. For a prediction query <inline-formula><mml:math id="ieqn65"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>r</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>, when the agent selects an action, <inline-formula><mml:math id="ieqn66"><mml:mi>&#x0394;</mml:mi><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> represents the distance between the query visit number <inline-formula><mml:math id="ieqn67"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and the current visit number <inline-formula><mml:math id="ieqn68"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, which is then encoded using an activation function. Thus, the representation of a diagnosis at the current step is given by <xref ref-type="disp-formula" rid="E4">Equation 4</xref>:</p><disp-formula id="E4"><label>(4)</label><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msubsup><mml:mi>e</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>w</mml:mi><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn69"><mml:mi>w</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn70"><mml:mi>b</mml:mi></mml:math></inline-formula> are learnable parameters, <inline-formula><mml:math id="ieqn71"><mml:mi>&#x03C3;</mml:mi></mml:math></inline-formula> is an activation function, and <inline-formula><mml:math id="ieqn72"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>;</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> denotes the vector concatenation operation.</p></sec><sec id="s2-5-4-2"><title>Dual-Path Encoding Module</title><p>The dual-path encoding module consists of the LSTM historical diagnosis encoding and the multilayer perceptron (MLP) nonhistorical diagnosis encoding.</p><sec id="s2-5-4-2-1"><title>LSTM Historical Diagnosis Encoding</title><p>For the historical diagnosis information for the patient, an LSTM layer encodes such information as the historical path embedding. The historical diagnosis of the patient is represented by <xref ref-type="disp-formula" rid="E5">Equation 5</xref>:</p><disp-formula id="E5"><label>(5)</label><mml:math id="eqn5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>h</mml:mi><mml:mi>n</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn73"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the action selected by the agent at step <inline-formula><mml:math id="ieqn74"><mml:mi>n</mml:mi></mml:math></inline-formula>, that is <inline-formula><mml:math id="ieqn75"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. Thus, <inline-formula><mml:math id="ieqn76"><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the selected action sequence before step <inline-formula><mml:math id="ieqn77"><mml:mi>n</mml:mi></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn78"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> denotes the first action taken from the initial state.</p><p>The term <inline-formula><mml:math id="ieqn79"><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is embedded into a continuous vector space <inline-formula><mml:math id="ieqn80"><mml:msup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn><mml:mi>d</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> then encoded. The historical path embedding is given by <xref ref-type="disp-formula" rid="E6">Equation 6</xref>:</p><disp-formula id="E6"><label>(6)</label><mml:math id="eqn6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>L</mml:mi><mml:mi>S</mml:mi><mml:mi>T</mml:mi><mml:mi>M</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>E</mml:mi><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn81"><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the embedding of <inline-formula><mml:math id="ieqn82"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn83"><mml:msubsup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mtext>LSTM</mml:mtext><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mi>E</mml:mi><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>. The superscript <inline-formula><mml:math id="ieqn84"><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:math></inline-formula> denotes the hidden representation produced by the historical LSTM branch.</p></sec><sec id="s2-5-4-2-2"><title>MLP Nonhistorical Diagnosis Encoding</title><p>For the nonhistorical diagnosis information for the patient, an MLP extracts latent semantic features of the nonhistorical diagnosis entities as nonhistorical path embedding. The representation of the nonhistorical diagnosis information for the patient consists of a fully connected layer and an activation function, as given by <xref ref-type="disp-formula" rid="E7">Equation 7</xref>:</p><disp-formula id="E7"><label>(7)</label><mml:math id="eqn7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msubsup><mml:mi>h</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi><mml:mi>h</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>L</mml:mi><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>;</mml:mo><mml:mi>p</mml:mi><mml:mo>;</mml:mo><mml:mi>E</mml:mi><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>W</mml:mi><mml:mn>2</mml:mn><mml:mo>&#x22C5;</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mn>1</mml:mn><mml:mo>&#x22C5;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>;</mml:mo><mml:mi>p</mml:mi><mml:mo>;</mml:mo><mml:mi>E</mml:mi><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn85"><mml:mi>W</mml:mi><mml:mn>1</mml:mn></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn86"><mml:mi>W</mml:mi><mml:mn>2</mml:mn></mml:math></inline-formula>, <inline-formula><mml:math id="ieqn87"><mml:mi>b</mml:mi><mml:mn>1</mml:mn></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn88"><mml:mi>b</mml:mi><mml:mn>2</mml:mn></mml:math></inline-formula> are learnable parameters, <inline-formula><mml:math id="ieqn89"><mml:mi>E</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the embedding of <inline-formula><mml:math id="ieqn90"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, and <inline-formula><mml:math id="ieqn91"><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>;</mml:mo></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula> denotes the vector concatenation operation. <inline-formula><mml:math id="ieqn92"><mml:mi>s</mml:mi></mml:math></inline-formula> and <inline-formula><mml:math id="ieqn93"><mml:mi>p</mml:mi></mml:math></inline-formula> denote the subject entity embedding and relation embedding of the current query, respectively.</p></sec></sec></sec></sec><sec id="s2-6"><title>Action Scoring Module</title><p>The policy network generates scores for each optional action and calculates the state transition probabilities. This module computes state transition probabilities jointly based on the outputs of the dual-path encoding module.</p><p>To balance the influence of historical and nonhistorical diagnoses, the event balance factor <inline-formula><mml:math id="ieqn94"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> was introduced to perform weighted allocation on their action scores. Ultimately, the probability distribution <inline-formula><mml:math id="ieqn95"><mml:mi>P</mml:mi></mml:math></inline-formula> over candidate actions is derived according to <xref ref-type="disp-formula" rid="E8">Equation 8</xref>:</p><disp-formula id="E8"><label>(8)</label><mml:math id="eqn8"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>his</mml:mtext></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mtext>nhis</mml:mtext></mml:mrow></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn96"><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mtext>his</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is the state transition probability distribution of historical diagnoses and <inline-formula><mml:math id="ieqn97"><mml:msub><mml:mrow><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mtext>nhis</mml:mtext></mml:mrow></mml:msub></mml:math></inline-formula> is the state transition probability distribution of nonhistorical diagnoses.</p></sec><sec id="s2-7"><title>Dynamic Weight Allocator Module</title><p>In the dynamic weight allocator module, the event balance factor <inline-formula><mml:math id="ieqn98"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> dynamically and interpretably adjusts the weight allocation in both the dual-channel aggregated reward function and the action scoring module.</p><p>The query <inline-formula><mml:math id="ieqn99"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi>q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> is encoded as <inline-formula><mml:math id="ieqn100"><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn101"><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the embedding vector of the query entity, <inline-formula><mml:math id="ieqn102"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the embedding vector of the query relation, and <inline-formula><mml:math id="ieqn103"><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the embedding vector of the visit number. The term <inline-formula><mml:math id="ieqn104"><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is input into a fully connected layer, and the weight <inline-formula><mml:math id="ieqn105"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> is generated through activation by the sigmoid function, as given by <xref ref-type="disp-formula" rid="E9">Equation 9</xref>:</p><disp-formula id="E9"><label>(9)</label><mml:math id="eqn9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>d</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>&#x22C5;</mml:mo><mml:msub><mml:mi>V</mml:mi><mml:mrow><mml:mi>q</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn106"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> denotes a query-level scalar output rather than an independent trainable parameter. The learnable parameters of the dynamic weight allocator are denoted by <inline-formula><mml:math id="ieqn107"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi>&#x03D5;</mml:mi><mml:mo>=</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>, and these parameters are optimized together with the policy network parameters.</p><p>When the query <inline-formula><mml:math id="ieqn108"><mml:mi>q</mml:mi></mml:math></inline-formula> focuses on historical diagnoses, <inline-formula><mml:math id="ieqn109"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> tends toward <inline-formula><mml:math id="ieqn110"><mml:mn>1</mml:mn></mml:math></inline-formula>; conversely, when the query <inline-formula><mml:math id="ieqn111"><mml:mi>q</mml:mi></mml:math></inline-formula> focuses on nonhistorical diagnoses, <inline-formula><mml:math id="ieqn112"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> tends toward <inline-formula><mml:math id="ieqn113"><mml:mn>0</mml:mn></mml:math></inline-formula>. The event balancing factor <inline-formula><mml:math id="ieqn114"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> enables the model to adaptively and interpretably adjust the decision-making basis according to the specific conditions of the current patient.</p></sec><sec id="s2-8"><title>Training Procedure</title><p>The dynamic weight allocator parameters <inline-formula><mml:math id="ieqn115"><mml:mi>&#x03D5;</mml:mi></mml:math></inline-formula> that generate <inline-formula><mml:math id="ieqn116"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> are optimized jointly with the policy network parameters <inline-formula><mml:math id="ieqn117"><mml:mi>&#x03B8;</mml:mi></mml:math></inline-formula>. The dual-channel aggregated reward $R$ is maximized via the REINFORCE algorithm [<xref ref-type="bibr" rid="ref39">39</xref>], such that the jointly trained policy network can be expressed as <inline-formula><mml:math id="ieqn118"><mml:msub><mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula>.</p><p>With the search space length fixed as <inline-formula><mml:math id="ieqn119"><mml:mi>L</mml:mi></mml:math></inline-formula>, the joint training objective of <inline-formula><mml:math id="ieqn120"><mml:mi>&#x03D5;</mml:mi></mml:math></inline-formula> and the policy network are optimized by maximizing the expected reward over the training sample set <inline-formula><mml:math id="ieqn121"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as given by <xref ref-type="disp-formula" rid="E10">Equation 10</xref>:</p><disp-formula id="E10"><label>(10)</label><mml:math id="eqn10"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mi>o</mml:mi><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2208;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mtext>train</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:munder><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>a</mml:mi><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub><mml:mo>&#x223C;</mml:mo><mml:mi>&#x03C0;</mml:mi></mml:mrow></mml:munder><mml:mi>R</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn122"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the training quadruple set, <inline-formula><mml:math id="ieqn123"><mml:mi>L</mml:mi></mml:math></inline-formula> denotes the maximum reasoning path length, and <inline-formula><mml:math id="ieqn124"><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> denotes the sampled action trajectory drawn from the jointly trained policy <inline-formula><mml:math id="ieqn125"><mml:msub><mml:mrow><mml:mi>&#x03C0;</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p>Optimization is performed with the REINFORCE algorithm by iterating over all quadruples in <inline-formula><mml:math id="ieqn126"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and updating the parameters through the following stochastic gradient, as given by <xref ref-type="disp-formula" rid="E11">Equation 11</xref>:</p><disp-formula id="E11"><label>(11)</label><mml:math id="eqn11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi mathvariant="normal">&#x2207;</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:msub><mml:mi>&#x03C0;</mml:mi><mml:mrow><mml:mi>&#x03B8;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>a</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x22C5;</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <italic>a</italic> denotes the action sampled at the current reasoning step and <inline-formula><mml:math id="ieqn127"><mml:mi>s</mml:mi></mml:math></inline-formula> denotes the corresponding state.</p><p>In the action scoring module, the query-level <inline-formula><mml:math id="ieqn128"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> participates in the policy distribution and therefore affects the policy gradient update through <inline-formula><mml:math id="ieqn129"><mml:mi>&#x03D5;</mml:mi></mml:math></inline-formula>. In the reward aggregation module, the generated <inline-formula><mml:math id="ieqn130"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> is substituted for <inline-formula><mml:math id="ieqn131"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> as a query-level weighting coefficient and is treated as a fixed numeric value when computing <inline-formula><mml:math id="ieqn132"><mml:mi>R</mml:mi></mml:math></inline-formula>; no separate gradient is taken through the reward calculation itself. The RL4TKGR training and inference pseudocode is provided in the RL4TKGR Training and Inference Pseudocode section in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-9"><title>Dataset Description</title><p>The CG-TKG dataset used in this study was derived from the clinical diagnosis and treatment data for patients with chronic gastritis treated by a chief physician at Guang&#x2019;anmen Hospital, China Academy of Chinese Medical Sciences, from March 2009 to October 2022. The dataset includes 17,906 patients with a total of 38,360 visit records (the visit number per patient ranges from 1 to 35). The source data were retrospective real-world clinical diagnosis and treatment records and included structured and semistructured information on symptoms, TCM diagnoses, Western medicine diagnoses, TCM syndrome diagnoses, laboratory tests, and gastroscopy and pathological examinations, which were used to construct temporal medical entities, relations, and visit-level time stamps in the CG-TKG. Before the dataset was provided to the research team, the source hospital and data center completed deidentification by removing direct patient identifiers and replacing original hospital identifiers with study-specific anonymous codes. The research team received and analyzed only the deidentified dataset. Because all records were obtained from a single institution and one chief physician&#x2019;s clinical practice, the constructed CG-TKG may reflect institution-specific and practitioner-specific diagnostic and prescribing patterns. The data cover the aspects described in the following paragraphs.</p><p>For symptoms and TCM diagnoses, 40,993 types of information were included, including chief complaints; present illness history; and TCM inspection, auscultation-olfaction, inquiry, and palpation (eg, gastric distension, stomach pain, poor appetite, heartburn, acid regurgitation, white tongue coating, slippery pulse).</p><p>Regarding Western medicine diagnoses, the information included 119 types of chronic gastritis diseases, including chronic atrophic gastritis, chronic superficial gastritis, erosive gastritis, and <italic>Helicobacter pylori</italic>&#x2013;associated gastritis.</p><p>TCM syndrome diagnoses included 263 types of results, such as spleen deficiency with dampness-heat syndrome, liver-stomach disharmony syndrome, spleen-stomach disharmony syndrome, liver depression and spleen deficiency syndrome, and spleen-stomach weakness syndrome.</p><p>There were a total of 59 laboratory test items such as comprehensive biochemistry, routine urine, complete blood cell analysis, routine stool, and occult blood involving 139 indicators such as white blood cells, red blood cells, urine protein, carbohydrate antigen 724, and alpha-fetoprotein and their results.</p><p>Gastroscopy and pathological examinations included 35 items (eg, active inflammation, intraepithelial neoplasia, atrophy, gastric mucosa texture, gastric mucosa color) and 56 descriptive indicators (eg, mild, moderate, normal, smooth and soft, rough mucosa).</p><p>Additional details on CG-TKG construction and quality control are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Briefly, entities and relations were extracted from structured and semistructured clinical fields, and each patient visit was treated as a visit-level temporal snapshot. Multisource observations recorded within the same visit share the same visit-order time stamp. Missing clinical items were not imputed nor treated as negative facts; only observed clinical facts were retained in the graph. The entity and relation schema was fully reviewed by TCM experts, and generated graph facts were further checked through sampled expert verification.</p><p>The CG-TKG dataset was partitioned by visit number. We selected patient data spanning 4, 8, and 12 visits to construct 3 temporal subsets. For each subset, the first 3, 7, and 11 visits (from cases with 4, 8, and 12 total visits, respectively) served as the training set, while the remaining visits were assigned to the validation and test sets. After manual verification by TCM experts, 3 dynamic chronic gastritis clinical datasets were obtained: CG-TKG-4, CG-TKG-8, and CG-TKG-12. <xref ref-type="table" rid="table1">Table 1</xref> presents the statistics of the CG-TKG dataset.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Chronic gastritis temporal knowledge graph (CG-TKG) dataset statistics.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Dataset</td><td align="left" valign="bottom">Entity count, n</td><td align="left" valign="bottom">Relation count, n</td><td align="left" valign="bottom">Training set size, n</td><td align="left" valign="bottom">Validation set size, n</td><td align="left" valign="bottom">Test set size, n</td><td align="left" valign="bottom">Time stamps</td><td align="left" valign="bottom" colspan="2">Time stamp interval, mean (variance)</td><td align="left" valign="bottom">Patient count, n</td></tr></thead><tbody><tr><td align="left" valign="top">CG-TKG-4</td><td align="left" valign="top">2304</td><td align="left" valign="top">132</td><td align="left" valign="top">161,697</td><td align="left" valign="top">27,206</td><td align="left" valign="top">27,206</td><td align="left" valign="top">4</td><td align="left" valign="top" colspan="2">37.00 (84.94)</td><td align="left" valign="top">772</td></tr><tr><td align="left" valign="top">CG-TKG-8</td><td align="left" valign="top">1225</td><td align="left" valign="top">116</td><td align="left" valign="top">68,357</td><td align="left" valign="top">4797</td><td align="left" valign="top">4797</td><td align="left" valign="top">8</td><td align="left" valign="top" colspan="2">38.06 (88.90)</td><td align="left" valign="top">140</td></tr><tr><td align="left" valign="top">CG-TKG-12</td><td align="left" valign="top">868</td><td align="left" valign="top">39</td><td align="left" valign="top">36,569</td><td align="left" valign="top">3031</td><td align="left" valign="top">3031</td><td align="left" valign="top">12</td><td align="left" valign="top" colspan="2">30.45 (62.25)</td><td align="left" valign="top">54</td></tr></tbody></table></table-wrap><p>The 3 temporal subsets were defined by visit course length and therefore have imbalanced patient counts: 772 patients in CG-TKG-4, 140 patients in CG-TKG-8, and 54 patients in CG-TKG-12. In particular, CG-TKG-12 represented a small long-course subgroup; therefore, results from this subset should be interpreted cautiously and require validation in larger long-term follow-up cohorts.</p></sec><sec id="s2-10"><title>Experimental Settings and Evaluation Metrics</title><p>RL4TKGR was implemented in PyTorch and initialized with uniform embeddings. The main hyperparameters were set as follows: The entity embedding dimension was 100, the relation embedding dimension was 100, the visit number embedding dimension was 20, the Adam optimizer was used for parameter optimization, the learning rate was 0.001, and the batch size was 512. All comparative and ablation experiments used the same dataset partitions and evaluation protocol described in the following paragraphs. To improve reproducibility without substantially expanding the main text, the source code and implementation scripts are publicly available through the project GitHub page [<xref ref-type="bibr" rid="ref40">40</xref>]. The public repository provides the executable training and evaluation scripts and the implementation-level settings not expanded in the main text, including optimizer arguments, regularization options, dropout configuration, LSTM configuration, checkpointing, random-seed handling, and evaluation scripts.</p><p>The MRR and Hits@1/3/10 were used to evaluate the performance of RL4TKGR in TKG prediction tasks. The formulas and explanations of these evaluation metrics are explained in the following sections.</p><sec id="s2-10-1"><title>MRR</title><p>The MRR is defined as the average of the reciprocals of the ranks of the first relevant result for all queries, as shown in <xref ref-type="disp-formula" rid="E12">Equation 12</xref>:</p><disp-formula id="E12"><label>(12)</label><mml:math id="eqn12"><mml:mtext>MRR</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mrow><mml:msubsup><mml:mo stretchy="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mtext>rank</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mrow></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn133"><mml:mi>Q</mml:mi></mml:math></inline-formula> is the set of queries and <inline-formula><mml:math id="ieqn134"><mml:msub><mml:mrow><mml:mtext>rank</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the rank of the first correct answer in the <inline-formula><mml:math id="ieqn135"><mml:mi>i</mml:mi></mml:math></inline-formula>-th query.</p></sec><sec id="s2-10-2"><title>Hits@k</title><p>Hits@k is defined as the hit rate of whether the correct answer appears in the top k results, as shown in <xref ref-type="disp-formula" rid="E13">Equation 13</xref>:</p><disp-formula id="E13"><label>(13)</label><mml:math id="eqn13"><mml:mtext>Hits@k</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mrow><mml:msubsup><mml:mo stretchy="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mtext>rank</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2264;</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mrow></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn136"><mml:mn>1</mml:mn><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x00B7;</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> is an indicator function that takes <inline-formula><mml:math id="ieqn137"><mml:mn>1</mml:mn></mml:math></inline-formula> if the condition in the parentheses is satisfied and <inline-formula><mml:math id="ieqn138"><mml:mn>0</mml:mn></mml:math></inline-formula> otherwise; <inline-formula><mml:math id="ieqn139"><mml:msub><mml:mrow><mml:mtext>rank</mml:mtext></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the rank of the first correct answer in the <inline-formula><mml:math id="ieqn140"><mml:mi>i</mml:mi></mml:math></inline-formula>-th query.</p><p>All comparative and ablation results were calculated as mean (SD) over 5 independent runs with different random initializations; the SD therefore reflects run-to-run variability. To keep the main manuscript concise, we summarized the comparative results as mean values only in the manuscript, and the full comparative mean (SD) results are provided in Tables S3-S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. The 95% CI was calculated as mean <inline-formula><mml:math id="ieqn141"><mml:mo>&#x00B1;</mml:mo><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mn>0.975,4</mml:mn></mml:mrow></mml:msub><mml:mo>&#x00D7;</mml:mo><mml:mtext>SD</mml:mtext><mml:mtext>/</mml:mtext><mml:msqrt><mml:mn>5</mml:mn></mml:msqrt></mml:math></inline-formula>, where <inline-formula><mml:math id="ieqn142"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:msub><mml:mi>t</mml:mi><mml:mrow><mml:mn>0.975</mml:mn><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>2.776</mml:mn></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>. For each comparator, an overall <italic>P</italic> value was calculated using a paired <italic>t</italic> test over matched repeated-run results across MRR and Hits@1/3/10. In the comparative experiments, RL4TKGR served as the reference model. In the ablation experiments, the full RL4TKGR model served as the reference model. The reported <italic>P</italic> values are row-level overall comparisons rather than metric-specific tests.</p><p>Because MRR and Hits@k are ranking-based model metrics calculated over test queries and averaged across 5 runs, they are reported as metric values rather than participant proportions. The test query counts were 27,206 for CG-TKG-4, 4797 for CG-TKG-8, and 3031 for CG-TKG-12, as shown in <xref ref-type="table" rid="table1">Table 1</xref>.</p><p>For link prediction evaluation, we used a time-aware filtered setting rather than a raw or static filtered setting. Specifically, for each test query <inline-formula><mml:math id="ieqn143"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>, candidate entities <inline-formula><mml:math id="ieqn144"><mml:msup><mml:mrow><mml:mi>o</mml:mi></mml:mrow><mml:mrow><mml:mi>`</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> that form known true quadruples <inline-formula><mml:math id="ieqn145"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>p</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>o</mml:mi></mml:mrow><mml:mrow><mml:mi>`</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math></inline-formula> with the same subject, relation, and time stamp in the training, validation, or test split were removed from the ranking list, except for the target answer. True quadruples occurring at different time stamps were retained because static filtering incorrectly removes temporally valid alternatives and is therefore unsuitable for TKG reasoning.</p><p>The maximum reasoning path length was fixed at <inline-formula><mml:math id="ieqn146"><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:math></inline-formula> for all experiments. This setting follows the TITer-style path-based reinforcement learning paradigm for TKG reasoning [<xref ref-type="bibr" rid="ref19">19</xref>], in which a small fixed number of hops is used to control search noise and computational cost. It also matched the visit-level CG-TKG setting because a 3-hop trajectory can traverse the main historical diagnosis or prescription chain while avoiding unnecessarily long paths to weakly related nodes.</p></sec></sec><sec id="s2-11"><title>Computational Cost and Convergence</title><p>All experiments were conducted on a workstation with one NVIDIA GeForce RTX 4090 GPU (24 GB), an Intel Core i7 CPU, and 16&#x2010;32 GB system RAM. With a batch size of 512, CG-TKG-4 contains approximately 316 batches per epoch and requires approximately 1.5 minutes to 3 minutes per epoch, with a total training time of approximately 2 hours to 4 hours. CG-TKG-8 contains approximately 134 batches per epoch and requires approximately 40 seconds to 80 seconds per epoch, with a total training time of approximately 1 hour to 2 hours. CG-TKG-12 contains approximately 72 batches per epoch and requires approximately 25 seconds to 45 seconds per epoch, with a total training time of approximately 0.5 hour to 1 hour. The peak GPU memory usage is approximately 2 GB to 4 GB for CG-TKG-4 and 2 GB to 3 GB for CG-TKG-8 and CG-TKG-12; peak system memory usage is approximately 4 GB to 8 GB, 3 GB to 6 GB, and 3 GB to 5 GB, respectively. Training uses a maximum of 50&#x2010;100 epochs with early stopping based on validation MRR and a patience of 5&#x2010;10 epochs. Models typically converge within approximately 40&#x2010;70 epochs, and CG-TKG-12 converges earlier than CG-TKG-4 because it contains fewer training quadruples and a more stable long-course visit structure. Based on the <xref ref-type="table" rid="table1">Table 1</xref> test-set sizes (27,206, 4797, and 3031 queries for CG-TKG-4, CG-TKG-8, and CG-TKG-12, respectively) and an average path-based reinforcement learning inference latency of approximately 2 ms per query on the same RTX 4090 GPU, test-set inference is estimated to require approximately 54 seconds, 10 seconds, and 6 seconds, respectively. The relatively small patient count in CG-TKG-12 may increase the risk of overfitting to long-course patient trajectories. To reduce this risk, we used separate training, validation, and test splits, validation-MRR-based early stopping, 5 independent runs with different random initializations, and comparative and ablation analyses. Nevertheless, external validation on larger and more balanced cohorts is needed.</p></sec><sec id="s2-12"><title>Introduction to Baseline Models</title><p>Here, we present the baseline models used for comparison. Static knowledge graph reasoning models included TransE [<xref ref-type="bibr" rid="ref14">14</xref>], relational graph convolutional network (R-GCN) [<xref ref-type="bibr" rid="ref35">35</xref>], structure-aware convolutional network (SACN) [<xref ref-type="bibr" rid="ref36">36</xref>], and DistMult [<xref ref-type="bibr" rid="ref16">16</xref>], while the TKG reasoning models included RE-GCN [<xref ref-type="bibr" rid="ref19">19</xref>], CEN [<xref ref-type="bibr" rid="ref27">27</xref>], RE-Net [<xref ref-type="bibr" rid="ref28">28</xref>], CENET [<xref ref-type="bibr" rid="ref29">29</xref>], and TITer [<xref ref-type="bibr" rid="ref26">26</xref>], as well as 3 recent TKG reasoning baselines: TiRGN [<xref ref-type="bibr" rid="ref27">27</xref>], RETIA [<xref ref-type="bibr" rid="ref28">28</xref>], and DaeMon [<xref ref-type="bibr" rid="ref29">29</xref>]. For fairness, all baseline models were configured with their optimal hyperparameters as reported in the original papers or tuned to achieve their best performance. Detailed descriptions of the baseline models are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Comparative Experiments</title><p>RL4TKGR was compared with the models listed in the Baseline Models section. The static knowledge graph models were retained as conventional reference baselines, but the primary fairness comparison was based on TKG reasoning models that can use temporal information. All baseline models used optimal performance parameters. The comparative experimental results are shown in <xref ref-type="table" rid="table2">Table 2</xref>. To keep the main manuscript concise, <xref ref-type="table" rid="table2">Table 2</xref> reports mean values only in a compact cross-dataset column format; the full comparative results with SDs and row-level paired <italic>t</italic> test <italic>P</italic> values are provided in Tables S3-S6 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparative experimental results.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="4">CG-TKG<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>-4</td><td align="left" valign="bottom" colspan="4">CG-TKG-8</td><td align="left" valign="bottom" colspan="4">CG-TKG-12</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">MRR<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td><td align="left" valign="bottom">MRR</td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td><td align="left" valign="bottom">MRR</td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td></tr></thead><tbody><tr><td align="left" valign="top">TransE</td><td align="left" valign="top">24.34</td><td align="left" valign="top">1.17</td><td align="left" valign="top">31.13</td><td align="left" valign="top">39.41</td><td align="left" valign="top">23.60</td><td align="left" valign="top">1.89</td><td align="left" valign="top">29.45</td><td align="left" valign="top">37.59</td><td align="left" valign="top">24.92</td><td align="left" valign="top">10.44</td><td align="left" valign="top">32.36</td><td align="left" valign="top">36.08</td></tr><tr><td align="left" valign="top">R-GCN<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup></td><td align="left" valign="top">13.39</td><td align="left" valign="top">6.21</td><td align="left" valign="top">14.51</td><td align="left" valign="top">27.34</td><td align="left" valign="top">15.72</td><td align="left" valign="top">7.21</td><td align="left" valign="top">15.33</td><td align="left" valign="top">28.51</td><td align="left" valign="top">14.15</td><td align="left" valign="top">6.31</td><td align="left" valign="top">14.25</td><td align="left" valign="top">31.17</td></tr><tr><td align="left" valign="top">SACN<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">20.28</td><td align="left" valign="top">11.80</td><td align="left" valign="top">20.99</td><td align="left" valign="top">36.95</td><td align="left" valign="top">22.91</td><td align="left" valign="top">18.76</td><td align="left" valign="top">27.13</td><td align="left" valign="top">39.35</td><td align="left" valign="top">21.89</td><td align="left" valign="top">9.93</td><td align="left" valign="top">25.27</td><td align="left" valign="top">39.08</td></tr><tr><td align="left" valign="top">DistMult</td><td align="left" valign="top">26.83</td><td align="left" valign="top">14.69</td><td align="left" valign="top">30.48</td><td align="left" valign="top">53.50</td><td align="left" valign="top">25.80</td><td align="left" valign="top">16.06</td><td align="left" valign="top">30.81</td><td align="left" valign="top">49.80</td><td align="left" valign="top">29.88</td><td align="left" valign="top">22.12</td><td align="left" valign="top">38.07</td><td align="left" valign="top">45.77</td></tr><tr><td align="left" valign="top">CEN<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">29.55</td><td align="left" valign="top">19.51</td><td align="left" valign="top">32.88</td><td align="left" valign="top">47.69</td><td align="left" valign="top">30.40</td><td align="left" valign="top">18.94</td><td align="left" valign="top">34.49</td><td align="left" valign="top">50.06</td><td align="left" valign="top">31.19</td><td align="left" valign="top">20.91</td><td align="left" valign="top">36.71</td><td align="left" valign="top">51.02</td></tr><tr><td align="left" valign="top">RE-GCN<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">30.98</td><td align="left" valign="top">21.60</td><td align="left" valign="top">34.50</td><td align="left" valign="top">49.22</td><td align="left" valign="top">31.55</td><td align="left" valign="top">20.84</td><td align="left" valign="top">35.69</td><td align="left" valign="top">51.19</td><td align="left" valign="top">32.13</td><td align="left" valign="top">23.34</td><td align="left" valign="top">38.23</td><td align="left" valign="top">53.59</td></tr><tr><td align="left" valign="top">RE-Net<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">31.89</td><td align="left" valign="top">22.40</td><td align="left" valign="top">34.55</td><td align="left" valign="top">50.17</td><td align="left" valign="top">32.49</td><td align="left" valign="top">21.07</td><td align="left" valign="top">37.45</td><td align="left" valign="top">53.09</td><td align="left" valign="top">35.38</td><td align="left" valign="top">25.67</td><td align="left" valign="top">38.90</td><td align="left" valign="top">55.07</td></tr><tr><td align="left" valign="top">CENET<sup><xref ref-type="table-fn" rid="table2fn8">h</xref></sup></td><td align="left" valign="top">35.11</td><td align="left" valign="top">29.45</td><td align="left" valign="top">50.71</td><td align="left" valign="top">59.37</td><td align="left" valign="top">37.69</td><td align="left" valign="top">30.96</td><td align="left" valign="top">50.39</td><td align="left" valign="top">60.25</td><td align="left" valign="top">40.32</td><td align="left" valign="top">31.42</td><td align="left" valign="top">52.69</td><td align="left" valign="top">63.92</td></tr><tr><td align="left" valign="top">TITer<sup><xref ref-type="table-fn" rid="table2fn9">i</xref></sup></td><td align="left" valign="top">34.02</td><td align="left" valign="top">31.48</td><td align="left" valign="top">35.40</td><td align="left" valign="top">39.98</td><td align="left" valign="top">50.56</td><td align="left" valign="top">47.97<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">52.64</td><td align="left" valign="top">53.93</td><td align="left" valign="top">49.55</td><td align="left" valign="top">42.05</td><td align="left" valign="top">56.57</td><td align="left" valign="top">62.36</td></tr><tr><td align="left" valign="top">TiRGN<sup><xref ref-type="table-fn" rid="table2fn11">k</xref></sup></td><td align="left" valign="top">37.32</td><td align="left" valign="top">32.47</td><td align="left" valign="top">38.53</td><td align="left" valign="top">41.97</td><td align="left" valign="top">50.92</td><td align="left" valign="top">41.10<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">53.12</td><td align="left" valign="top">55.93</td><td align="left" valign="top">50.51</td><td align="left" valign="top">43.31</td><td align="left" valign="top">56.82</td><td align="left" valign="top">63.15</td></tr><tr><td align="left" valign="top">RETIA<sup><xref ref-type="table-fn" rid="table2fn13">m</xref></sup></td><td align="left" valign="top">42.12</td><td align="left" valign="top">33.83</td><td align="left" valign="top">40.65</td><td align="left" valign="top">48.43</td><td align="left" valign="top">51.14</td><td align="left" valign="top">45.57<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">56.93</td><td align="left" valign="top">57.14</td><td align="left" valign="top">53.76</td><td align="left" valign="top">44.95</td><td align="left" valign="top">59.07</td><td align="left" valign="top">67.28</td></tr><tr><td align="left" valign="top">DaeMon<sup><xref ref-type="table-fn" rid="table2fn14">n</xref></sup></td><td align="left" valign="top">43.26</td><td align="left" valign="top">34.41</td><td align="left" valign="top">43.90</td><td align="left" valign="top">51.54</td><td align="left" valign="top">51.76</td><td align="left" valign="top">46.60<sup><xref ref-type="table-fn" rid="table2fn12">l</xref></sup></td><td align="left" valign="top">57.45</td><td align="left" valign="top">63.94</td><td align="left" valign="top">54.87</td><td align="left" valign="top">45.27</td><td align="left" valign="top">60.11</td><td align="left" valign="top">68.95</td></tr><tr><td align="left" valign="top">RL4TKGR<sup><xref ref-type="table-fn" rid="table2fn15">o</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table2fn16">p</xref></sup></td><td align="left" valign="top">51.66<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">35.39<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">53.14<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">60.15<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">53.18<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">45.48</td><td align="left" valign="top">59.52<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">67.26<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">56.95<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">49.75<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">62.54<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td><td align="left" valign="top">71.79<sup><xref ref-type="table-fn" rid="table2fn10">j</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>CG-TKG: chronic gastritis temporal knowledge graph.</p></fn><fn id="table2fn2"><p><sup>b</sup>MRR: mean reciprocal rank.</p></fn><fn id="table2fn3"><p><sup>c</sup>R-GCN: relational graph convolutional network.</p></fn><fn id="table2fn4"><p><sup>d</sup>SACN: structure-aware convolutional network.</p></fn><fn id="table2fn5"><p><sup>e</sup>CEN: complex evolutional network.</p></fn><fn id="table2fn6"><p><sup>f</sup>RE-GCN: relation-enhanced graph convolutional network.</p></fn><fn id="table2fn7"><p><sup>g</sup>RE-Net: recurrent event network.</p></fn><fn id="table2fn8"><p><sup>h</sup>CENET: contrastive event network.</p></fn><fn id="table2fn9"><p><sup>i</sup>TITer: TimeTraveler.</p></fn><fn id="table2fn10"><p><sup>j</sup>Best result in the column.</p></fn><fn id="table2fn11"><p><sup>k</sup>TiGRN: time-guided recurrent graph network.</p></fn><fn id="table2fn12"><p><sup>l</sup>Second-best result in the column.</p></fn><fn id="table2fn13"><p><sup>m</sup>RETIA: relation-entity twin-interact aggregation.</p></fn><fn id="table2fn14"><p><sup>n</sup>DaeMon: adaptive path-memory network.</p></fn><fn id="table2fn15"><p><sup>o</sup>RL4TKGR: reinforcement learning&#x2013;based temporal knowledge graph reasoning.</p></fn><fn id="table2fn16"><p><sup>p</sup>Reference model.</p></fn></table-wrap-foot></table-wrap><p><xref ref-type="table" rid="table2">Table 2</xref> shows the comparison of prediction performance between RL4TKGR and various static knowledge graph reasoning models and TKG reasoning models on the 3 temporal datasets (CG-TKG-4/8/12). H1, H3, and H10 denote Hits@1, Hits@3, and Hits@10, respectively. The results from the overall paired <italic>t</italic> tests across MRR and Hits@1/3/10 were significantly different between RL4TKGR and all baseline models on CG-TKG-4 and CG-TKG-12 (<italic>P</italic>&#x003C;.001 for all comparisons). For CG-TKG-8, the differences were also significant for all baselines (<italic>P</italic>&#x003C;.001), except for TITer, for which the overall paired <italic>t</italic> test yielded a <italic>P</italic>=.004. For the 3 newly recent baselines, the comparisons between RL4TKGR and TiRGN, RETIA, and DaeMon all yielded <italic>P</italic>&#x003C;.001 on CG-TKG-4, CG-TKG-8, and CG-TKG-12. These results indicate that temporal modeling is important for analyzing the evolutionary process of chronic gastritis, while the remaining Hits@1 advantage of TITer on CG-TKG-8 is reported as a metric-specific exception rather than as evidence of a specific mechanism.</p><p>RL4TKGR achieved the highest MRR, Hits@3, and Hits@10 on all 3 datasets and the highest Hits@1 on CG-TKG-4 and CG-TKG-12. DaeMon had the strongest MRR at baseline on all 3 subsets. Compared with DaeMon, RL4TKGR increased MRR from 43.26 (SD 0.95; 95% CI 42.08&#x2010;44.44) to 51.66 (SD 0.76; 95% CI 50.72&#x2010;52.60) on CG-TKG-4, from 51.76 (SD 1.12; 95% CI 50.37&#x2010;53.15) to 53.18 (SD 0.79; 95% CI 52.20&#x2010;54.16) on CG-TKG-8, and from 54.87 (SD 1.17; 95% CI 53.42&#x2010;56.32) to 56.95 (SD 0.84; 95% CI 55.91&#x2010;57.99) on CG-TKG-12. For Hits@10, RL4TKGR also outperformed DaeMon by reaching MRRs of 60.15 (SD 0.91) versus 51.54 (SD 1.08) on CG-TKG-4, 67.26 (SD 0.96) versus 63.94 (SD 1.31) on CG-TKG-8, and 71.79 (SD 1.03) versus 68.95 (SD 1.39) on CG-TKG-12. The only first-rank exception was Hits@1 on CG-TKG-8, where TITer reached 47.97 (SD 1.18), DaeMon reached 46.60 (SD 1.04), and RL4TKGR reached 45.48 (SD 0.71). Because Hits@1 requires the correct answer to be ranked first, we interpreted this as a metric-specific exception rather than as evidence that RL4TKGR was uniformly superior on every ranking criterion. Overall, however, the updated comparison indicated that RL4TKGR retains a consistent advantage over recent TKG reasoning models in MRR, Hits@3, and Hits@10, while the static baselines were retained only as conventional reference models.</p></sec><sec id="s3-2"><title>Ablation Study</title><p>Ablation experiments were designed to verify the effectiveness of the historical diagnosis encoding module and historical reward in RL4TKGR. The experiments removed the historical diagnosis encoding module (w/o e), removed the historical reward (w/o r), removed both simultaneously (w/o e&#x0026;r), and evaluated their performance on the 3 datasets CG-TKG-4, CG-TKG-8, and CG-TKG-12. <xref ref-type="table" rid="table3">Table 3</xref> reports the experimental results as mean (SD) over 5 runs. The full model resulted in significant overall differences from all ablated versions in paired <italic>t</italic> tests across MRR and Hits@1/3/10 (<italic>P</italic>&#x003C;.001 for all comparisons).</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Ablation study results.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="4">CG-TKG<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup>-4</td><td align="left" valign="bottom" colspan="4">CG-TKG-8</td><td align="left" valign="bottom" colspan="4">CG-TKG-12</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">MRR<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td><td align="left" valign="bottom">MRR</td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td><td align="left" valign="bottom">MRR</td><td align="left" valign="bottom">Hits@1</td><td align="left" valign="bottom">Hits@3</td><td align="left" valign="bottom">Hits@10</td></tr></thead><tbody><tr><td align="left" valign="top">w/o e<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">44.05</td><td align="left" valign="top">33.85</td><td align="left" valign="top">47.65</td><td align="left" valign="top">55.34</td><td align="left" valign="top">50.53</td><td align="left" valign="top">41.91</td><td align="left" valign="top">55.95</td><td align="left" valign="top">64.86</td><td align="left" valign="top">54.30</td><td align="left" valign="top">46.43</td><td align="left" valign="top">63.54<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">68.32</td></tr><tr><td align="left" valign="top">w/o r<sup><xref ref-type="table-fn" rid="table3fn5">e</xref></sup></td><td align="left" valign="top">41.59</td><td align="left" valign="top">30.18</td><td align="left" valign="top">45.93</td><td align="left" valign="top">52.18</td><td align="left" valign="top">49.55</td><td align="left" valign="top">40.04</td><td align="left" valign="top">54.56</td><td align="left" valign="top">62.35</td><td align="left" valign="top">53.16</td><td align="left" valign="top">44.72</td><td align="left" valign="top">60.88</td><td align="left" valign="top">66.39</td></tr><tr><td align="left" valign="top">w/o e&#x0026;r<sup><xref ref-type="table-fn" rid="table3fn6">f</xref></sup></td><td align="left" valign="top">33.54</td><td align="left" valign="top">28.75</td><td align="left" valign="top">39.12</td><td align="left" valign="top">45.89</td><td align="left" valign="top">34.49</td><td align="left" valign="top">31.52</td><td align="left" valign="top">36.56</td><td align="left" valign="top">39.52</td><td align="left" valign="top">42.95</td><td align="left" valign="top">36.51</td><td align="left" valign="top">44.76</td><td align="left" valign="top">57.66</td></tr><tr><td align="left" valign="top">RL4TKGR<sup><xref ref-type="table-fn" rid="table3fn7">g</xref></sup></td><td align="left" valign="top">51.66<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">35.39<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">53.14<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">60.15<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">53.18<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">45.48<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">59.52<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">67.26<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">56.95<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">49.75<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td><td align="left" valign="top">62.54</td><td align="left" valign="top">71.79<sup><xref ref-type="table-fn" rid="table3fn4">d</xref></sup></td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>CG-TKG: chronic gastritis temporal knowledge graph.</p></fn><fn id="table3fn2"><p><sup>b</sup>MRR: mean reciprocal rank.</p></fn><fn id="table3fn3"><p><sup>c</sup>w/o e: historical diagnosis encoding module removed.</p></fn><fn id="table3fn4"><p><sup>d</sup>Best result in the column.</p></fn><fn id="table3fn5"><p><sup>e</sup>w/o r historical reward removed.</p></fn><fn id="table3fn6"><p><sup>f</sup>w/o e&#x0026;r: historical diagnosis encoding module and historical rewarad simultaneously removed.</p></fn><fn id="table3fn7"><p><sup>g</sup>RL4TKGR: reinforcement learning&#x2013;based temporal knowledge graph reasoning.</p></fn></table-wrap-foot></table-wrap><p>Specifically, on the CG-TKG-4 dataset, the MRR of the full model was 51.66 (SD 0.76) compared with 44.05 (SD 0.82) for w/o e, 41.59 (SD 0.88) for w/o r, and 33.54 (SD 0.74) for w/o e&#x0026;r. This pattern was also observed on CG-TKG-8 and CG-TKG-12 for MRR, Hits@1, and Hits@10. For Hits@3 on CG-TKG-12, however, w/o e reached an MRR of 63.54 (SD 0.98), which was higher than the full model value of 62.54 (SD 0.93). This exception indicated that the historical diagnosis encoding module does not monotonically improve every individual ranking metric. Nevertheless, the full RL4TKGR model retained higher MRR, Hits@1, and Hits@10 on CG-TKG-12 and had the best overall ablation profile across the 3 datasets.</p><p>Comparing the impact of ablating a single module, the removal of the historical diagnosis encoding module (w/o e) caused slightly more damage to performance than the removal of the historical reward (w/o r). This indicated that the diagnosis encoding module played a more fundamental role in modeling patients&#x2019; historical states. Nevertheless, when both modules were ablated simultaneously (w/o e&#x0026;r), the performance exhibited a steep decline, which suggested that the historical reward substantially enhanced the effectiveness of the diagnosis encoding module and that the 2 components demonstrated a strong complementary relationship.</p></sec><sec id="s3-3"><title>Experiments on the Effectiveness of the Event Balancing Factor in the Reward Function</title><p>In the RL4TKGR model, the reward function calculated the weighted value of the historical reward and the nonhistorical reward using a hyperparameter &#x03B1;, where &#x03B1; was provided by the value of the event balance factor &#x03B2;. To verify the impact of the event balance factor &#x03B2; in the reward function on model performance, experiments were conducted on 3 different datasets: CG-TKG-4, CG-TKG-8, and CG-TKG-12. The hyperparameter &#x03B1; in the reward function was fixed to guide the learning of the policy network, and a hyperparameter search was performed on each dataset with &#x03B1; ranging from 0 to 1. The experimental results are shown in <xref ref-type="fig" rid="figure2">Figure 2</xref>, where the 3 curves represent the fixed-<inline-formula><mml:math id="ieqn147"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> MRR sensitivity results on CG-TKG-4, CG-TKG-8, and CG-TKG-12, respectively. On CG-TKG-4, when <inline-formula><mml:math id="ieqn148"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> was fixed, the MRR value ranged from 45.0 to 51.0, while the MRR value of RL4TKGR was 51.66. On CG-TKG-8 and CG-TKG-12, the best fixed-<inline-formula><mml:math id="ieqn149"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> settings appeared around 0.55 and 0.8, respectively. These descriptive trends suggested that longer visit histories placed greater weight on historical diagnostic information. Because no paired <italic>t</italic> test <italic>P</italic> value was calculated for the fixed-<inline-formula><mml:math id="ieqn150"><mml:mi>&#x03B1;</mml:mi></mml:math></inline-formula> hyperparameter sweep, these results are reported as descriptive sensitivity analyses rather than formal significance tests.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Effectiveness experiment of &#x03B2; on chronic gastritis temporal knowledge graph (CG-TKG)-4, CG-TKG-8, and CG-TKG-12. MRR: mean reciprocal rank.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e83544_fig02.png"/></fig><p>In summary, the application of the event balance factor <inline-formula><mml:math id="ieqn151"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> in the reward function had a consistent descriptive effect on model performance. In particular, when trained jointly with the policy network, <inline-formula><mml:math id="ieqn152"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> could adaptively adjust the weights of historical and nonhistorical information, thereby optimizing the overall performance of the model. Therefore, dynamically adjusting <inline-formula><mml:math id="ieqn153"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> was an effective means of improving model performance.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Model Interpretability Analysis</title><p>In the dynamic weight allocator module, the event balance factor &#x03B2; dynamically and interpretably adjusted the weight distribution between the dual-channel aggregation reward function and the action scoring module. When the query focused on historical diagnosis information, &#x03B2; approached 1; when the query focused on nonhistorical diagnosis information, &#x03B2; approaches 0.</p><p>The factor &#x03B2; enabled the model to automatically and interpretably adjust the source weights of decision-making bases according to the patient&#x2019;s specific current condition. In the design of the dual-channel aggregation reward function, the value of &#x03B1; was directly provided by &#x03B2;. As shown in the hyperparameter experimental results in <xref ref-type="fig" rid="figure2">Figure 2</xref>, the interpretability was explained as follows: (1) On CG-TKG-4, in which the maximum number of patient visits was 4, the probability of disease progression was relatively high, whereas historical correlations were weak. The model attained its highest MRR when &#x03B1; was between 0.4 and 0.5, providing evidence of its interpretability. (2) On CG-TKG-8, in which the maximum number of patient visits was 8, the disease state tended to stabilize, the likelihood of progression decreased, and historical correlations became stronger. The model achieved its highest MRR when &#x03B1; was between 0.5 and 0.6. (3) On CG-TKG-12, where the maximum number of patient visits was twelve, the disease state was the most stable, the likelihood of progression was the lowest, and historical correlations were the strongest. The model attained its highest MRR when &#x03B1; is approximately 0.8.</p><p>To further examine patient-level interpretability, we analyzed 1686 query-level <inline-formula><mml:math id="ieqn154"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> values exported during model inference. As shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>, the mean <inline-formula><mml:math id="ieqn155"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> increased with visit order across all 3 datasets, from 0.413 to 0.462 in CG-TKG-4, from 0.400 to 0.557 in CG-TKG-8, and from 0.393 to 0.763 in CG-TKG-12. Because larger <inline-formula><mml:math id="ieqn156"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> values indicate greater reliance on the historical diagnostic path, this pattern indicated that RL4TKGR progressively increased the weight assigned to accumulated historical information as longitudinal visit information became richer. In the illustrative anonymized case, Patient A had the same direction of change: <inline-formula><mml:math id="ieqn157"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> increased from 0.41 to 0.69 for disease prediction and from 0.45 to 0.73 for medication prediction between visits 2 and 4. These patient-level trajectories provided a direct visualization of how the model dynamically shifted from nonhistorical or current visit information toward history-weighted reasoning.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Patient-level interpretability of the event balance factor <inline-formula><mml:math id="ieqn158"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula>: (A) mean (SD in shaded areas) query-level <inline-formula><mml:math id="ieqn159"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> values across visit order for chronic gastritis temporal knowledge graph (CG-TKG)-4, CG-TKG-8, and CG-TKG-12 and (B) <inline-formula><mml:math id="ieqn160"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> trajectory of the illustrative anonymized case (Patient A) for disease and medication prediction queries. Values &#x003E;0.5 indicate history-weighted reasoning, whereas values &#x003C;0.5 indicate greater reliance on nonhistorical or current-visit information.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e83544_fig03.png"/></fig></sec><sec id="s4-2"><title>Path-Level and Counterfactual Interpretability</title><p>The event balance factor <inline-formula><mml:math id="ieqn161"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> should not be interpreted as a conventional clinical attention weight. Instead, it is a query-level model weight that quantifies how strongly the policy relies on historical versus nonhistorical reasoning paths. To move beyond a single top-ranked case result, we exported the actual reasoning paths for Patient A and provide them in Tables S13-S15 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. For the disease diagnosis query at the third visit, RL4TKGR predicted chronic gastritis with <inline-formula><mml:math id="ieqn162"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.58</mml:mn></mml:math></inline-formula> and followed the historical diagnostic chain Patient A <inline-formula><mml:math id="ieqn163"><mml:mo>&#x2192;</mml:mo></mml:math></inline-formula> chronic gastritis at visit 1 <inline-formula><mml:math id="ieqn164"><mml:mo>&#x2192;</mml:mo></mml:math></inline-formula> chronic gastritis at visit 2 <inline-formula><mml:math id="ieqn165"><mml:mo>&#x2192;</mml:mo></mml:math></inline-formula> the third-visit diagnosis query. For the medication query, RL4TKGR predicted Zhizhu Kuanzhong capsules with <inline-formula><mml:math id="ieqn166"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.62</mml:mn></mml:math></inline-formula> and followed the previous prescription path across the first 2 visits.</p><p>The same exported paths also showed the nonhistorical component of the prediction. For the disease diagnosis query, the current visit symptom path linked abdominal distension, poor belching, and phlegm in the pharynx to chronic gastritis, with a nonhistorical contribution of <inline-formula><mml:math id="ieqn167"><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.42</mml:mn></mml:math></inline-formula>. For the medication query, the current visit symptom path linked abdominal distension and loose stools to Zhizhu Kuanzhong capsules, with a nonhistorical contribution of <inline-formula><mml:math id="ieqn168"><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0.38</mml:mn></mml:math></inline-formula>. As a patient-level counterfactual check, when the historical path was masked by forcing <inline-formula><mml:math id="ieqn169"><mml:mi>&#x03B2;</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula>, the true disease diagnosis fell from rank 1 to rank 4, and the true medication fell from rank 1 to rank 5. These results indicated that the historical reasoning path contributed to the rank-1 predictions rather than merely accompanying them.</p><p>At the cohort level, the ablation study provided an additional counterfactual view of interpretability. Removing the historical diagnosis encoding module or the historical reward substantially lowered MRR across the CG-TKG datasets, indicating that the historical path components were not decorative but materially contributed to model performance. Thus, the interpretability evidence in RL4TKGR consisted of query-level <inline-formula><mml:math id="ieqn170"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> weighting, exported path-level evidence, and counterfactual changes after masking or removing historical-path information.</p></sec><sec id="s4-3"><title>Case Analysis of Chronic Gastritis Diagnosis Prediction</title><p>We selected 3 visits by an anonymized case (Patient A) as examples for analysis. Patient A was a study-specific anonymous label and did not correspond to any original hospital identifier. To reduce the risk of reidentification, the case is reported using visit order rather than calendar dates. This case corresponds to the patient-level <inline-formula><mml:math id="ieqn171"><mml:mi>&#x03B2;</mml:mi></mml:math></inline-formula> trajectory shown in <xref ref-type="fig" rid="figure3">Figure 3B</xref>, providing a link between the model&#x2019;s dynamic weighting mechanism and the case-level prediction results provided in the following paragraphs.</p><p>In the first visit, the manifestations included occasional stomach pain after acute gastroenteritis, abdominal distension, poor belching, a foreign body sensation in the pharynx, phlegm in the pharynx, normal appetite, average sleep quality with many dreams, and normal urination and defecation. The tongue was dark red with a thin white coating, and the pulse was deep and thready. The diagnosis was chronic gastritis, and the prescribed medicines were Zhizhu Kuanzhong capsules and Qingyan tablets.</p><p>In the second visit, the manifestations were disappearance of stomach pain, occasional abdominal distension, a foreign body sensation in the pharynx, poor belching, phlegm in the pharynx, normal appetite, average sleep quality, improvement in dreaminess, and normal urination and defecation. The tongue was dark red with a thin white coating that was slightly yellowish, and the pulse was deep and thready. The diagnosis was chronic gastritis, and the prescribed medicines were Zhizhu Kuanzhong capsules and Qingyan Tablets.</p><p>In the third visit, the manifestations were loose stools in the past week, occasional abdominal distension, poor belching, phlegm in the pharynx, normal appetite, average sleep quality with many dreams, and normal urination. The tongue was dark red with a thin white coating, and the pulse was deep and thready.</p><p>Using RL4TKGR, based on the historical diagnostic information for Patient A and the symptom information from the third visit, the top 10 probabilistic reasoning results for the anonymized third-visit queries (Patient A, prescribe medicine, ?, 3) and (Patient A, disease diagnosis, ?, 3) were generated. As shown in <xref ref-type="table" rid="table4">Table 4</xref>, the actual results from the third visit for Patient A are shown among the other results. The analysis showed that, in both query cases, RL4TKGR ranked the correct result first in the prediction sequence, suggesting that the model could rank plausible candidate diagnoses and medications for clinician review while preserving links to temporal medical entities.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Case reasoning results.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Ranking</td><td align="left" valign="bottom">(Patient 49375, prescribe medicine, ?, 3), predicted entities</td><td align="left" valign="bottom">(Patient 49375, disease diagnosis, ?, 3), predicted entities</td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">Zhizhu Kuanzhong capsules<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">Chronic gastritis<bold><sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></bold></td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">Guben Yichang tablets<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">Chronic atrophic gastritis</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">Simo Decoction oral liquid</td><td align="left" valign="top">Chronic superficial gastritis</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">Bazhen Yimu pills</td><td align="left" valign="top">Erosive gastritis</td></tr><tr><td align="left" valign="top">5</td><td align="left" valign="top">Qingyan tablets</td><td align="left" valign="top"><italic>Helicobacter pylori</italic>&#x2013;associated gastritis</td></tr><tr><td align="left" valign="top">6</td><td align="left" valign="top">Xuhanting capsules</td><td align="left" valign="top">Spleen deficiency and dampness-heat syndrome</td></tr><tr><td align="left" valign="top">7</td><td align="left" valign="top">Wuweizi granules</td><td align="left" valign="top">Mild dysplasia of chronic gastritis</td></tr><tr><td align="left" valign="top">8</td><td align="left" valign="top">Qizhi Weitong granules</td><td align="left" valign="top">Chronic gastritis with intestinal metaplasia</td></tr><tr><td align="left" valign="top">9</td><td align="left" valign="top">Xuhan Weitong granules</td><td align="left" valign="top">Chronic gastritis with hiatal hernia</td></tr><tr><td align="left" valign="top">10</td><td align="left" valign="top">Jinghua Weikang capsules</td><td align="left" valign="top">Liver stagnation and spleen deficiency syndrome</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Actual results from the third visit for Patient A.</p></fn></table-wrap-foot></table-wrap><p>The subsequent follow-up results for Patient A, represented as the fourth visit in the anonymized sequence, indicated marked clinical improvement: alleviated abdominal distension, belching, reduced night sweats, general chills, disappearance of phlegm, normal appetite, average sleep, sometimes formed stools, and normal urination. The patient was diagnosed with chronic gastritis. The patient&#x2019;s reported symptoms improved after taking the prescribed drugs, and the further progression of chronic gastritis was controlled in this illustrative case. These results supported the plausibility of the predicted medication ranking and indicated that RL4TKGR may help predict future disease status and recommend candidate therapies for clinical review.</p></sec><sec id="s4-4"><title>Conclusion</title><p>We investigated the temporal dependency of chronic gastritis progression and, based on CG-TKG constructed from TCM diagnostic and treatment data, proposed RL4TKGR, a reinforcement learning&#x2013;based TKG reasoning model. By designing a reinforcement learning policy network that exploits the associations among historical diagnosis data, the model achieved interpretable prediction of chronic gastritis diagnosis and treatment. The experimental results support the effectiveness of RL4TKGR on the evaluated CG-TKG reasoning tasks.</p><p>RL4TKGR may provide model-level support for ranking candidate diagnoses and medications in longitudinal chronic gastritis care. Because the current CG-TKG is constructed from a TCM-specific symbolic entity space and has not been externally validated across datasets, institutions, or disease domains, its generalizability to other clinical settings remains to be established. In addition, because the data originate from one clinical setting and one chief physician&#x2019;s practice, the model&#x2019;s performance may not directly generalize to other hospitals, clinicians, regional practice styles, or patient populations without external validation. Future work should first evaluate whether the architecture can be adapted to other longitudinal clinical knowledge graphs through disease-specific schema design, external validation, and prospective clinical evaluation. From a translational perspective, RL4TKGR may serve as a prototype decision support tool that ranks candidate diagnoses and medications for clinician review in longitudinal chronic gastritis care. Future prospective validation and collaboration among clinicians, data governance teams, and machine learning researchers are needed before deployment in real-world clinical workflows.</p><p>Another limitation is that the current CG-TKG uses visit-order time stamps rather than continuous calendar time. Although <xref ref-type="table" rid="table1">Table 1</xref> reports the mean and variance of visit intervals, RL4TKGR does not explicitly use elapsed days between visits. Future work should incorporate time-gap embeddings, temporal decay functions, or continuous-time TKG models to capture heterogeneous follow-up intervals more precisely.</p></sec></sec></body><back><ack><p>GPT-5.5 was used solely for grammar and language checking during manuscript preparation. The authors take full responsibility for all content.</p></ack><notes><sec><title>Funding</title><p>This work is supported in part by the National Natural Science Foundation of China (number 82374621), the Research Project of China Academy of Chinese Medical Sciences "Evidence Study Based on Multimodal Knowledge Graph Reasoning of the Idea of Treating Pre-disease in TCM (2023016)", and the Fundamental Research Funds for the Central Public Welfare Research Institutes (number ZZ1718-XRZ-101-SJ).</p></sec><sec><title>Data Availability</title><p>The source code for implementing the reinforcement learning&#x2013;based temporal knowledge graph reasoning (RL4TKGR) is publicly available through the project GitHub page [<xref ref-type="bibr" rid="ref40">40</xref>]. The code release includes the model architecture, training and evaluation scripts, and implementation-level hyperparameter settings; therefore, detailed options such as optimizer arguments, weight decay or regularization choices, dropout settings, and long short-term memory (LSTM) hidden-layer configuration can be inspected directly in the released source code. The individual-level chronic gastritis temporal knowledge graph (CG-TKG) clinical dataset cannot be publicly released at this stage because it is derived from real-world clinical diagnosis and treatment records and is subject to project confidentiality, institutional ethics approval, and hospital data governance restrictions. Deidentified aggregate dataset statistics and model evaluation results are provided in the manuscript and <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. Any request for additional access to nonpublic data requires approval from the source institution and the relevant ethics and data governance bodies.</p></sec></notes><fn-group><fn fn-type="con"><p>Data curation: XQ, ZS, YW, RZ</p><p>Formal analysis: GS, RZ</p><p>Funding acquisition: DL, XZ</p><p>Investigation: XQ, ZS, YW</p><p>Methodology: XQ, DL</p><p>Software: HL</p><p>Validation: YW, HL</p><p>Visualization: XQ, HL</p><p>Writing &#x2013; original draft: XQ, ZS, LY</p><p>Writing &#x2013; review &#x0026; editing: XQ, DL</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AI</term><def><p>artificial intelligence</p></def></def-item><def-item><term id="abb2">CEN</term><def><p>complex evolutional network</p></def></def-item><def-item><term id="abb3">CENET</term><def><p>contrastive event network</p></def></def-item><def-item><term id="abb4">CG-TKG</term><def><p>chronic gastritis temporal knowledge graph</p></def></def-item><def-item><term id="abb5">DaeMon</term><def><p>adaptive path-memory network</p></def></def-item><def-item><term id="abb6">EHR</term><def><p>electronic health record</p></def></def-item><def-item><term id="abb7">EMR</term><def><p>electronic medical record</p></def></def-item><def-item><term id="abb8">GRU</term><def><p>gated recurrent unit</p></def></def-item><def-item><term id="abb9">LSTM</term><def><p>long short-term memory</p></def></def-item><def-item><term id="abb10">MLP</term><def><p>multilayer perceptron</p></def></def-item><def-item><term id="abb11">MRR</term><def><p>mean reciprocal rank</p></def></def-item><def-item><term id="abb12">R-GCN</term><def><p>relational graph convolutional network</p></def></def-item><def-item><term id="abb13">RE-GCN</term><def><p>relation-enhanced graph convolutional network</p></def></def-item><def-item><term id="abb14">RE-Net</term><def><p>recurrent event network</p></def></def-item><def-item><term id="abb15">RETIA</term><def><p>relation-entity twin-interact aggregation</p></def></def-item><def-item><term id="abb16">RL4TKGR</term><def><p>reinforcement learning&#x2013;based temporal knowledge graph reasoning</p></def></def-item><def-item><term id="abb17">RNN</term><def><p>recurrent neural network</p></def></def-item><def-item><term id="abb18">RTTI</term><def><p>run-time type information</p></def></def-item><def-item><term id="abb19">SACN</term><def><p>structure-aware convolutional network</p></def></def-item><def-item><term id="abb20">TANGO</term><def><p>Temporal Knowledge Graph Forecasting with Neural Ordinary Equations</p></def></def-item><def-item><term id="abb21">TCM</term><def><p>traditional Chinese medicine</p></def></def-item><def-item><term id="abb22">TiGRN</term><def><p>time-guided recurrent graph network</p></def></def-item><def-item><term id="abb23">TITer</term><def><p>TimeTraveler</p></def></def-item><def-item><term id="abb24">TKG</term><def><p>temporal knowledge graph</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Banks</surname><given-names>M</given-names> </name><name name-style="western"><surname>Graham</surname><given-names>D</given-names> </name><name name-style="western"><surname>Jansen</surname><given-names>M</given-names> </name><etal/></person-group><article-title>British Society of Gastroenterology guidelines on the diagnosis and management of patients at risk of gastric adenocarcinoma</article-title><source>Gut</source><year>2019</year><month>09</month><volume>68</volume><issue>9</issue><fpage>1545</fpage><lpage>1575</lpage><pub-id pub-id-type="doi">10.1136/gutjnl-2018-318126</pub-id><pub-id pub-id-type="medline">31278206</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>F</given-names> </name><name name-style="western"><surname>Yuan</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Diagnosing chronic atrophic gastritis by gastroscopy using artificial intelligence</article-title><source>Dig Liver Dis</source><year>2020</year><month>05</month><volume>52</volume><issue>5</issue><fpage>566</fpage><lpage>572</lpage><pub-id pub-id-type="doi">10.1016/j.dld.2019.12.146</pub-id><pub-id pub-id-type="medline">32061504</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Turtoi</surname><given-names>DC</given-names> </name><name name-style="western"><surname>Brata</surname><given-names>VD</given-names> </name><name name-style="western"><surname>Incze</surname><given-names>V</given-names> </name><etal/></person-group><article-title>Artificial intelligence for the automatic diagnosis of gastritis: a systematic review</article-title><source>J Clin Med</source><year>2024</year><month>08</month><day>15</day><volume>13</volume><issue>16</issue><fpage>4818</fpage><pub-id pub-id-type="doi">10.3390/jcm13164818</pub-id><pub-id pub-id-type="medline">39200959</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>N</given-names> </name><name name-style="western"><surname>Liao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Peng</surname><given-names>L</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>C</given-names> </name></person-group><article-title>A deep learning method to assist with chronic atrophic gastritis diagnosis using white light images</article-title><source>Dig Liver Dis</source><year>2022</year><month>11</month><volume>54</volume><issue>11</issue><fpage>1513</fpage><lpage>1519</lpage><pub-id pub-id-type="doi">10.1016/j.dld.2022.04.025</pub-id><pub-id pub-id-type="medline">35610166</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tao</surname><given-names>X</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>An artificial intelligence system for chronic atrophic gastritis diagnosis and risk stratification under white light endoscopy</article-title><source>Dig Liver Dis</source><year>2024</year><month>08</month><volume>56</volume><issue>8</issue><fpage>1319</fpage><lpage>1326</lpage><pub-id pub-id-type="doi">10.1016/j.dld.2024.01.177</pub-id><pub-id pub-id-type="medline">38246825</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Lim</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Seo</surname><given-names>SI</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>J</given-names> </name><name name-style="western"><surname>Shin</surname><given-names>W</given-names> </name></person-group><article-title>A deep recurrent neural network-based explainable prediction model for progression from atrophic gastritis to gastric cancer</article-title><source>Applied Sciences</source><year>2021</year><volume>11</volume><issue>13</issue><fpage>6194</fpage><pub-id pub-id-type="doi">10.3390/app11136194</pub-id><pub-id pub-id-type="medline">36003951</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>ES</given-names> </name><name name-style="western"><surname>Mudiganti</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Risk of gastric adenocarcinoma in a multiethnic population undergoing routine care: an electronic health records cohort study</article-title><source>Cancer Epidemiol Biomarkers Prev</source><year>2024</year><month>04</month><day>3</day><volume>33</volume><issue>4</issue><fpage>547</fpage><lpage>556</lpage><pub-id pub-id-type="doi">10.1158/1055-9965.EPI-23-1200</pub-id><pub-id pub-id-type="medline">38231023</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Towards time-aware knowledge graph completion</article-title><conf-name>26th International Conference on Computational Linguistics</conf-name><conf-date>Dec 11-16, 2016</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/C16-1.pdf">https://aclanthology.org/C16-1.pdf</ext-link></comment></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Kuang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Explainable AI method for tinnitus diagnosis via neighbor-augmented knowledge graph and traditional Chinese medicine: development and validation study</article-title><source>JMIR Med Inform</source><year>2024</year><month>06</month><day>10</day><volume>12</volume><fpage>e57678</fpage><pub-id pub-id-type="doi">10.2196/57678</pub-id><pub-id pub-id-type="medline">38857077</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Song</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Tang</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Research of medical aided diagnosis system based on temporal knowledge graph</article-title><conf-name>Advanced Data Mining and Applications: 16th International Conference</conf-name><conf-date>Nov 12-14, 2020</conf-date><pub-id pub-id-type="doi">10.1007/978-3-030-65390-3_19</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bourque</surname><given-names>VR</given-names> </name><name name-style="western"><surname>Poulain</surname><given-names>C</given-names> </name><name name-style="western"><surname>Proulx</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Genetic and phenotypic similarity across major psychiatric disorders: a systematic review and quantitative assessment</article-title><source>Transl Psychiatry</source><year>2024</year><month>03</month><day>30</day><volume>14</volume><issue>1</issue><fpage>171</fpage><pub-id pub-id-type="doi">10.1038/s41398-024-02866-3</pub-id><pub-id pub-id-type="medline">38555309</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>J</given-names> </name></person-group><article-title>A survey on temporal knowledge graphs-extrapolation and interpolation tasks</article-title><conf-name>The International Conference on Natural Computation, Fuzzy Systems and Knowledge Discovery</conf-name><conf-date>Jul 30 to Aug 1, 2022</conf-date><pub-id pub-id-type="doi">10.1007/978-3-031-20738-9_110</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>T</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>T</given-names> </name><name name-style="western"><surname>Ge</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Encoding temporal information for time-aware link prediction</article-title><conf-name>2016 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 1-4, 2016</conf-date><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/D16-1">http://aclweb.org/anthology/D16-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D16-1260</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Bordes</surname><given-names>A</given-names> </name><name name-style="western"><surname>Usunier</surname><given-names>N</given-names> </name><name name-style="western"><surname>Garcia-Duran</surname><given-names>A</given-names> </name><etal/></person-group><article-title>Translating embeddings for modeling multi-relational data</article-title><conf-name>NIPS&#x2019;13: 27th International Conference on Neural Information Processing Systems</conf-name><conf-date>Dec 5-10, 2013</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/10.5555/2999792.2999923">https://dl.acm.org/doi/10.5555/2999792.2999923</ext-link></comment></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Garc&#x00ED;a-Dur&#x00E1;n</surname><given-names>A</given-names> </name><name name-style="western"><surname>Duman&#x010D;i&#x0107;</surname><given-names>S</given-names> </name><name name-style="western"><surname>Niepert</surname><given-names>M</given-names> </name></person-group><article-title>Learning sequence encoders for temporal knowledge graph completion</article-title><conf-name>2018 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Oct 31 to Nov 4, 2018</conf-date><comment><ext-link ext-link-type="uri" xlink:href="http://aclweb.org/anthology/D18-1">http://aclweb.org/anthology/D18-1</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/D18-1516</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Yih</surname><given-names>SW</given-names> </name><name name-style="western"><surname>He</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Embedding entities and relations for learning and inference in knowledge bases</article-title><conf-name>International Conference on Learning Representations (ICLR) 2015</conf-name><conf-date>May 7-9, 2015</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.microsoft.com/en-us/research/publication/embedding-entities-and-relations-for-learning-and-inference-in-knowledge-bases/">https://www.microsoft.com/en-us/research/publication/embedding-entities-and-relations-for-learning-and-inference-in-knowledge-bases/</ext-link></comment></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Goel</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kazemi</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Brubaker</surname><given-names>M</given-names> </name><name name-style="western"><surname>Poupart</surname><given-names>P</given-names> </name></person-group><article-title>Diachronic embedding for temporal knowledge graph completion</article-title><source>AAAI</source><year>2020</year><volume>34</volume><issue>4</issue><fpage>3988</fpage><lpage>3995</lpage><pub-id pub-id-type="doi">10.1609/aaai.v34i04.5815</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>xERTE: explainable reasoning on temporal knowledge graphs for forecasting future links</article-title><source>arXiv</source><comment>Preprint posted online on December 31, 2020</comment><pub-id pub-id-type="doi">10.48550/arXiv.2012.15537</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>H</given-names> </name><name name-style="western"><surname>Zhong</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>He</surname><given-names>K</given-names> </name></person-group><article-title>TimeTraveler: reinforcement learning for temporal knowledge graph forecasting</article-title><conf-name>2021 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 7-11, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2021.emnlp-main">https://aclanthology.org/2021.emnlp-main</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.655</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>R</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>G</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Tian</surname><given-names>Y</given-names> </name></person-group><article-title>Reinforcement learning with time intervals for temporal knowledge graph reasoning</article-title><source>Inf Syst</source><year>2024</year><month>02</month><volume>120</volume><fpage>102292</fpage><pub-id pub-id-type="doi">10.1016/j.is.2023.102292</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>W</given-names> </name><etal/></person-group><article-title>Temporal knowledge graph reasoning based on evolutional representation learning</article-title><conf-name>SIGIR &#x2019;21</conf-name><conf-date>Jul 11-15, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3404835">https://dl.acm.org/doi/proceedings/10.1145/3404835</ext-link></comment><pub-id pub-id-type="doi">10.1145/3404835.3462963</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Guan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>X</given-names> </name><etal/></person-group><article-title>Complex evolutional pattern learning for temporal knowledge graph reasoning</article-title><conf-name>60th Annual Meeting of the Association for Computational Linguistics</conf-name><conf-date>May 22-27, 2022</conf-date><comment>arXiv:220307782</comment><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2022.acl-short">https://aclanthology.org/2022.acl-short</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2022.acl-short.32</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Qu</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Recurrent event network: global structure inference over temporal knowledge graph</article-title><source>arXiv</source><comment>Preprint posted online on  Oct 6, 2020</comment><comment><ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1904.05530v3">https://arxiv.org/abs/1904.05530v3</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ou</surname><given-names>J</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>H</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>L</given-names> </name></person-group><article-title>Temporal knowledge graph reasoning with historical contrastive learning</article-title><source>AAAI</source><year>2023</year><volume>37</volume><issue>4</issue><fpage>4765</fpage><lpage>4773</lpage><pub-id pub-id-type="doi">10.1609/aaai.v37i4.25601</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Trivedi</surname><given-names>R</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Song</surname><given-names>L</given-names> </name></person-group><article-title>Know-evolve: deep temporal reasoning for dynamic knowledge graphs</article-title><conf-name>34th International Conference on Machine Learning</conf-name><conf-date>Aug 7-9, 2017</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v70/trivedi17a.html">https://proceedings.mlr.press/v70/trivedi17a.html</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Han</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Gu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Tresp</surname><given-names>V</given-names> </name></person-group><article-title>Learning neural ordinary equations for forecasting future links on temporal knowledge graphs</article-title><conf-name>2021 Conference on Empirical Methods in Natural Language Processing</conf-name><conf-date>Nov 7-11, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://aclanthology.org/2021.emnlp-main">https://aclanthology.org/2021.emnlp-main</ext-link></comment><pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.658</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>J</given-names> </name></person-group><article-title>TiRGN: time-guided recurrent graph network with local-global historical patterns for temporal knowledge graph reasoning</article-title><conf-name>Thirty-First International Joint Conference on Artificial Intelligence {IJCAI-22}</conf-name><conf-date>Jul 23-29, 2022</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ijcai.org/proceedings/2022">https://www.ijcai.org/proceedings/2022</ext-link></comment><pub-id pub-id-type="doi">10.24963/ijcai.2022/299</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>F</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>G</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>H</given-names> </name></person-group><article-title>RETIA: relation-entity twin-interact aggregation for temporal knowledge graph extrapolation</article-title><conf-name>2023 IEEE 39th International Conference on Data Engineering (ICDE)</conf-name><conf-date>Apr 3-7, 2023</conf-date><pub-id pub-id-type="doi">10.1109/ICDE55515.2023.00138</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Dong</surname><given-names>H</given-names> </name><name name-style="western"><surname>Ning</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>P</given-names> </name><etal/></person-group><article-title>Adaptive path-memory network for temporal knowledge graph reasoning</article-title><conf-name>Thirty-Second International Joint Conference on Artificial Intelligence {IJCAI-23}</conf-name><conf-date>Aug 19-25, 2023</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.ijcai.org/proceedings/2023">https://www.ijcai.org/proceedings/2023</ext-link></comment><pub-id pub-id-type="doi">10.24963/ijcai.2023/232</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Hildebrandt</surname><given-names>M</given-names> </name><name name-style="western"><surname>Joblin</surname><given-names>M</given-names> </name><name name-style="western"><surname>Tresp</surname><given-names>V</given-names> </name></person-group><article-title>TLogic: temporal logical rules for explainable link forecasting on temporal knowledge graphs</article-title><source>AAAI</source><year>2022</year><volume>36</volume><issue>4</issue><fpage>4120</fpage><lpage>4127</lpage><pub-id pub-id-type="doi">10.1609/aaai.v36i4.20330</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zheng</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Cai</surname><given-names>X</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>J</given-names> </name><name name-style="western"><surname>Su</surname><given-names>Q</given-names> </name></person-group><article-title>A review of knowledge graph applications in the medical field</article-title><source>Sheng Wu Yi Xue Gong Cheng Xue Za Zhi</source><year>2023</year><month>10</month><day>25</day><volume>40</volume><issue>5</issue><fpage>1040</fpage><lpage>1044</lpage><pub-id pub-id-type="doi">10.7507/1001-5515.202204016</pub-id><pub-id pub-id-type="medline">37879936</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Carvalho</surname><given-names>RM</given-names> </name><name name-style="western"><surname>Teixeira</surname><given-names>AS</given-names> </name><name name-style="western"><surname>Pesquita</surname><given-names>C</given-names> </name></person-group><article-title>Building a temporal knowledge graph for electronic health records</article-title><conf-name>DAO-XAI 2024: Workshop on Data meets Applied Ontologies in Explainable AI</conf-name><conf-date>Oct 19-20, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://ceur-ws.org/Vol-3833/paper6.pdf">https://ceur-ws.org/Vol-3833/paper6.pdf</ext-link></comment></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Geng</surname><given-names>S</given-names> </name><name name-style="western"><surname>Tao</surname><given-names>B</given-names> </name><name name-style="western"><surname>Liang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Fu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>W</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>B</given-names> </name></person-group><article-title>Temporal knowledge graph attention network for online doctor recommendation</article-title><conf-name>2023 8th International Conference on Intelligent Information Processing</conf-name><conf-date>Nov 21-22, 2023</conf-date><pub-id pub-id-type="doi">10.1145/3635175.3635224</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chaturvedi</surname><given-names>R</given-names> </name></person-group><article-title>Temporal knowledge graph extraction and modeling across multiple documents for health risk prediction</article-title><conf-name>WWW &#x2019;24</conf-name><conf-date>May 13-17, 2024</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/proceedings/10.1145/3589335">https://dl.acm.org/doi/proceedings/10.1145/3589335</ext-link></comment><pub-id pub-id-type="doi">10.1145/3589335.3651256</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Postiglione</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bean</surname><given-names>D</given-names> </name><name name-style="western"><surname>Kraljevic</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Dobson</surname><given-names>RJ</given-names> </name><name name-style="western"><surname>Moscato</surname><given-names>V</given-names> </name></person-group><article-title>Predicting future disorders via temporal knowledge graphs and medical ontologies</article-title><source>IEEE J Biomed Health Inform</source><year>2024</year><month>07</month><volume>28</volume><issue>7</issue><fpage>4238</fpage><lpage>4248</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2024.3390419</pub-id><pub-id pub-id-type="medline">38635388</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Lippman</surname><given-names>SA</given-names> </name></person-group><article-title>Dynamic programming and Markov decision processes</article-title><source>The New Palgrave Dictionary of Economics</source><year>2018</year><publisher-name>Palgrave Macmillan</publisher-name><fpage>3158</fpage><lpage>3164</lpage><pub-id pub-id-type="doi">10.1057/978-1-349-95189-5_80</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chaudhari</surname><given-names>PA</given-names> </name><name name-style="western"><surname>Khot</surname><given-names>SS</given-names> </name></person-group><article-title>Alzheimer&#x2019;s disease prediction using CAdam optimized reinforcement learning-based deep convolutional neural network model</article-title><source>Biomed Signal Process Control</source><year>2025</year><month>10</month><volume>108</volume><fpage>107968</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2025.107968</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Diao</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Optimizing vital signs in patients with traumatic brain injury: reinforcement learning algorithm development and validation</article-title><source>J Med Internet Res</source><year>2025</year><volume>27</volume><fpage>e63847</fpage><lpage>e63847</lpage><pub-id pub-id-type="doi">10.2196/63847</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Williams</surname><given-names>RJ</given-names> </name></person-group><article-title>Simple statistical gradient-following algorithms for connectionist reinforcement learning</article-title><source>Mach Learn</source><year>1992</year><month>05</month><volume>8</volume><issue>3-4</issue><fpage>229</fpage><lpage>256</lpage><pub-id pub-id-type="doi">10.1007/BF00992696</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="web"><article-title>QuXiaolong0812/RL4TKGR</article-title><source>GitHub</source><access-date>2026-08-22</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/QuXiaolong0812/RL4TKGR">https://github.com/QuXiaolong0812/RL4TKGR</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Baseline model descriptions, evaluation metrics and statistical reporting, full comparative results, full ablation results, fixed-alpha sensitivity analysis results, extended case analysis, model construction and quality control, statistical CIs and computational cost, specific case findings, terminology glossary, and model training and inference pseudocode.</p><media xlink:href="medinform_v14i1e83544_app1.docx" xlink:title="DOCX File, 587 KB"/></supplementary-material></app-group></back></article>