<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e78523</article-id><article-id pub-id-type="doi">10.2196/78523</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Automated Renal Tumor Segmentation in Computed Tomography Images Using a Global Attention&#x2013;Based DeepLabV3+ Model: Algorithm Development and Validation</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Yueyan</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Jianqiang</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shao</surname><given-names>Lingyu</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Lin</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Zhaoqing</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Liu</surname><given-names>Yujie</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Wen</surname><given-names>Jiaxin</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hao</surname><given-names>Xinyao</given-names></name><degrees>BSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names>Shuyan</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Jianhong</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Song</surname><given-names>Boming</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>School of Medical Information and Engineering, Xuzhou Medical University</institution><addr-line>KJL_E202 Room, 209 Tongshan Road, Yunlong District</addr-line><addr-line>Xuzhou</addr-line><addr-line>Jiangsu</addr-line><country>China</country></aff><aff id="aff2"><institution>Department of Radiology, The Second Hospital &#x0026; Clinical Medical School, Lanzhou University</institution><addr-line>Lanzhou</addr-line><addr-line>Gansu</addr-line><country>China</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>li</surname><given-names>dongliang</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>Ramin</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Liang</surname><given-names>Xiaolong</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Boming Song, PhD, School of Medical Information and Engineering, Xuzhou Medical University, KJL_E202 Room, 209 Tongshan Road, Xuzhou Medical University, Yunlong District, Xuzhou, Jiangsu, 221004, China, 86 15852036109; <email>boming.song@xzhmu.edu.cn</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>15</day><month>9</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e78523</elocation-id><history><date date-type="received"><day>05</day><month>06</month><year>2025</year></date><date date-type="rev-recd"><day>19</day><month>07</month><year>2026</year></date><date date-type="accepted"><day>21</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yueyan Zhao, Jianqiang Liu, Lingyu Shao, Lin Li, Zhaoqing Liu, Yujie Liu, Jiaxin Wen, Xinyao Hao, Shuyan Li, Jianhong Zhao, Boming Song. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 15.9.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e78523"/><abstract><sec><title>Background</title><p>The rising global incidence of renal tumors necessitates precise diagnostic interventions. Accurate segmentation of computed tomography (CT) scans is essential for nephron-sparing surgery and radiotherapy. However, conventional manual delineation is labor-intensive and prone to significant interobserver variability due to tumor morphological heterogeneity. There is an urgent clinical demand for robust, automated segmentation solutions.</p></sec><sec><title>Objective</title><p>This study aims to develop and validate GAM-DeepLabV3+, an automated framework designed to address boundary ambiguity and high false-positive rates in complex renal imaging scenarios.</p></sec><sec sec-type="methods"><title>Methods</title><p>We propose an optimized encoder-decoder architecture specifically tailored for renal mass detection. The framework incorporates three key innovations: (1) a lightweight MobileNetV2 backbone to minimize computational overhead for clinical deployment; (2) an Atrous Spatial Pyramid Pooling (ASPP) module to capture multiscale contextual information; and (3) a Global Attention Mechanism (GAM) in the decoder to enhance channel-spatial interactions, thereby refining boundary delineation by suppressing background noise. The model was rigorously evaluated on a private clinical dataset (n=218) and the KiTS19 benchmark (n=210).</p></sec><sec sec-type="results"><title>Results</title><p>GAM-DeepLabV3+ consistently outperformed state-of-the-art baselines. On the private dataset, the model achieved a mean Dice similarity coefficient (DSC) of 0.939 (SD 0.008), significantly surpassing feature pyramid network (FPN; mean 0.893, SD 0.008; <italic>P</italic>&#x003C;.001) and no-new-Net (nnU-Net; 2D, custom; mean 0.908, SD 0.013; <italic>P</italic>&#x003C;.001). It also achieved a mean 95% Hausdorff distance (HD95) of 1.485 (SD 0.522) pixels. On the KiTS19 dataset, it maintained a mean robust DSC of 0.928 (SD 0.006). To facilitate clinical translation, a demonstration-only online platform was developed.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The GAM-DeepLabV3+ framework provides an accurate, efficient, and fully automated solution for renal tumor segmentation. By overcoming boundary ambiguity and optimizing feature fusion, this approach shows potential as a decision-support aid, pending future validation with 3D reconstruction.</p></sec></abstract><kwd-group><kwd>kidney tumor segmentation</kwd><kwd>CT images</kwd><kwd>DeepLabV3+</kwd><kwd>global attention mechanism</kwd><kwd>deep learning</kwd><kwd>KiTS19</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The kidney is a parenchymal organ situated bilaterally along the spinal column beneath the costal arches, encapsulated by a thin layer of connective tissue and adipose tissue. It plays essential physiological roles in excreting metabolic waste products, maintaining water-electrolyte homeostasis, and secreting critical hormones, including erythropoietin and renin. Against the backdrop of a continuously rising global burden of malignant disease, kidney tumors [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>] have emerged as one of the most prevalent malignancies of the urinary system. Epidemiologically, their incidence ranks second only to bladder cancer among urological tumors. In 2020 alone, more than 430,000 new cases of kidney cancer were diagnosed worldwide, and the incidence continues to rise with advances in diagnostic technology and accelerating population aging, posing formidable challenges to kidney cancer prevention and treatment across diverse health care settings [<xref ref-type="bibr" rid="ref3">3</xref>]. The World Health Organization (WHO) has formally designated kidney tumors as a critical global public health priority [<xref ref-type="bibr" rid="ref4">4</xref>].</p><p>Kidney tumors encompass 3 principal histological categories [<xref ref-type="bibr" rid="ref5">5</xref>]: clear cell renal cell carcinoma (ccRCC), Wilms tumor, and transitional cell carcinoma of the renal pelvis. ccRCC, originating from the proximal tubular epithelium of the kidney, constitutes the most common primary renal malignancy in adults, accounting for approximately 80%&#x2010;85% of all cases. Wilms tumor predominantly affects pediatric populations, whereas transitional cell carcinoma of the renal pelvis arises at the pelvicalyceal junction and is characterized by urothelial differentiation. At the molecular pathological level, ccRCC&#x2014;the predominant subtype of renal cell carcinoma&#x2014;is closely associated with mutations in the von Hippel-Lindau (VHL) gene [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>]. In current clinical practice, computed tomography (CT) [<xref ref-type="bibr" rid="ref8">8</xref>], magnetic resonance imaging, and ultrasonography [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>] constitute the principal imaging modalities for evaluating renal lesions. Because most patients with kidney tumors lack specific early symptoms and are frequently diagnosed at an advanced stage, early detection and accurate imaging-based diagnosis are of paramount clinical importance. However, visual lesion detection alone is insufficient to guide complex therapeutic decision-making. Following confirmed or highly suspected diagnosis, precise 3D reconstruction [<xref ref-type="bibr" rid="ref11">11</xref>] and quantitative analysis of the tumor region are fundamental prerequisites for surgical planning, treatment response assessment, and prognostication. Yet conventional clinical workflows rely on manual tumor delineation by radiologists&#x2014;a process that is not only time-consuming and labor-intensive, but also susceptible to interobserver variability arising from subjective interpretation, potentially compromising the consistency and accuracy of treatment planning. Accordingly, the development of automated segmentation technology holds substantial significance for improving diagnostic efficiency, guiding surgical planning, and optimizing therapeutic strategies.</p><p>Early landmark contributions in this field focused primarily on enhancing feature extraction capability through architectural innovation. Yu et al [<xref ref-type="bibr" rid="ref12">12</xref>] proposed Crossbar-Net, which introduced crossbar-shaped patch sampling and a cascaded training strategy to strengthen the capture of local textural features, achieving a Dice similarity coefficient (DSC) of 0.91 on kidney tumor segmentation tasks. To address organ boundary ambiguity in arterial-phase 3D CT scans, Myronenko and Hatamizadeh [<xref ref-type="bibr" rid="ref13">13</xref>] designed a boundary-aware fully convolutional network that incorporated explicit edge supervision signals, elevating the joint kidney-tumor segmentation DSC to above 0.89 and establishing the importance of boundary constraints in fine-grained segmentation. This trajectory was further substantiated by studies in the KiTS challenge series: Heller et al [<xref ref-type="bibr" rid="ref14">14</xref>] and Zhao et al [<xref ref-type="bibr" rid="ref15">15</xref>] demonstrated, through ensemble optimization strategies and multiscale supervised U&#x2011;Net (MSS U-Net), respectively, that multiscale contextual information is critical for overcoming tumor morphological heterogeneity, with kidney and tumor DSC on the KiTS19 dataset consistently exceeding 0.97 and 0.80.</p><p>To address the persistent challenge of detecting small positive regions under limited supervision, subsequent research integrated residual connections with attention mechanisms. Guo et al [<xref ref-type="bibr" rid="ref16">16</xref>] proposed residual and attention U-Net (RAU-Net), and Zhao et al [<xref ref-type="bibr" rid="ref17">17</xref>] introduced boundary attention U-Net (BAU-Net), both of which used main-auxiliary branch synergy and weighted loss functions to substantially improve model sensitivity to small lesions. Notably, BAU-Net achieved kidney DSC of 0.98 and tumor DSC of 0.84 in the KiTS2021 challenge, marking the maturation of boundary attention mechanisms. More recent research has converged on 2 emerging directions. The first addresses the efficiency-interpretability trade-offs, exemplified by UNet-PWP developed by Rao et al [<xref ref-type="bibr" rid="ref18">18</xref>], which reduces computational complexity through pretrained weight transfer and adaptive partitioning while incorporating explainable AI to enhance clinical trustworthiness. Along this line, recent studies have incorporated Grad-CAM and attention-based visualization techniques within EfficientNetV2-based frameworks to provide interpretable segmentation results across both clinical and public datasets [<xref ref-type="bibr" rid="ref19">19</xref>]. Furthermore, combining Vision Transformers (ViTs) with conditional random fields (CRFs) has been shown to enhance boundary refinement while maintaining model transparency, offering a promising direction for explainable segmentation systems [<xref ref-type="bibr" rid="ref20">20</xref>]. The second direction centers on global context modeling, as demonstrated by the simplified UNETR developed by Choi et al [<xref ref-type="bibr" rid="ref21">21</xref>], which leverages the long-range dependency capture of Transformer architectures and integrates organ-level prior information for simultaneous segmentation, confirming that anatomical context integration significantly improves tumor localization accuracy [<xref ref-type="bibr" rid="ref22">22</xref>]. Recent advances have further explored hybrid architectures that combine convolutional neural networks (CNNs) with ViTs to leverage both local feature extraction and global context modeling. For instance, multiscale fusion strategies integrated with UNet and Transformer backbones have demonstrated superior performance in capturing heterogeneous tumor structures [<xref ref-type="bibr" rid="ref23">23</xref>]. In addition, cascaded Transformer frameworks incorporating mechanisms such as Gumbel-Softmax have been proposed to improve multiclass segmentation by enhancing feature selection and stage-wise refinement [<xref ref-type="bibr" rid="ref24">24</xref>]. Analogous findings in brain tumor segmentation have demonstrated that backbone selection critically influences detection accuracy [<xref ref-type="bibr" rid="ref25">25</xref>], and that integrating DeepLabV3+ with attention mechanisms yields consistent improvements across both public benchmarks and private clinical datasets [<xref ref-type="bibr" rid="ref26">26</xref>], further supporting the design rationale of this study.</p><p>Despite impressive benchmark performance reported by these methods [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>], their clinical translation remains substantially constrained. Prevailing high-performance approaches predominantly rely on 3D cascaded architectures or heavyweight Transformer models. Although 3D convolutions fully exploit volumetric spatial continuity, their substantial graphics processing unit (GPU) memory requirements and high computational overhead introduce considerable inference latency, precluding real-time deployment on standard clinical workstations or in time-critical emergency settings. Furthermore, complex cascaded pipelines increase system instability and maintenance burden. In response to this accuracy-efficiency dilemma, Lin et al [<xref ref-type="bibr" rid="ref29">29</xref>] proposed an efficient 2D segmentation framework designed to transcend the prevailing 3D paradigm by optimizing receptive field design and feature aggregation in 2D CNNs, thereby substantially reducing computational complexity while preserving high segmentation accuracy for small and morphologically complex kidney tumors.</p><p>Notwithstanding the accuracy benchmarks established by the aforementioned 3D and hybrid architectures, their high computational cost and complex deployment pipelines remain primary barriers to clinical translation. Specifically, existing mainstream methods exhibit several key limitations: first, the prohibitive GPU memory demands of 3D convolutions restrict real-time inference on standard medical hardware [<xref ref-type="bibr" rid="ref30">30</xref>], rendering them incompatible with the low-latency requirements of emergency care or intraoperative navigation; second, heavyweight Transformer models [<xref ref-type="bibr" rid="ref31">31</xref>] often require large-scale pretraining data and have been reported to be more susceptible to missed detections of small lesions under oversmoothing or receptive field mismatch [<xref ref-type="bibr" rid="ref32">32</xref>]; third, most methods fail to fully exploit deep semantic context at the 2D slice level during feature fusion, resulting in insufficiently sharp segmentation boundaries in challenging scenarios characterized by low tumor-to-background contrast and indistinct margins.</p><p>This study presents GAM-DeepLabV3+, a framework based on DeepLabV3+ for automated renal tumor segmentation (<xref ref-type="fig" rid="figure1">Figure 1</xref>), designed to reconcile the trade-off between accuracy and efficiency and to enhance the segmentation of small, low-contrast tumors. The system enhances the capture of semantic contextual information in tumor regions through backbone network refinement and the introduction of a global attention module. A targeted feature fusion strategy is further devised with the aim of improving tumor boundary delineation in complex backgrounds, including challenging cases such as small kidney tumors. The specific contributions of this work are as follows:</p><list list-type="order"><list-item><p>To reduce model parameter count, a lightweight enhanced MobileNetV2 is adopted as the backbone network, thereby strengthening the low-level feature extraction capability of GAM-DeepLabV3+.</p></list-item><list-item><p>Two Global Attention Mechanism (GAM) modules are introduced at key positions within the decoder: one is placed after a 1&#x00D7;1 convolution and prior to feature concatenation, while the other is embedded within the upsampling pathway. This configuration enables efficient fusion of deep and shallow features, enhancing the representation of salient information while suppressing background noise.</p></list-item><list-item><p>A BCEDiceLoss function is used to simultaneously mitigate class imbalance and refine boundary detail optimization. The proposed method achieves a mean Dice coefficient of 0.939 (SD 0.008) and 0.928 (SD 0.006) on a private dataset and the KiTS19 preoperative CT dataset, respectively, demonstrating its capacity for accurate renal tumor segmentation.</p></list-item></list><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>An overview of the complete technical pipeline and workflow of this study. CT: computed tomography; GAM: Global Attention Mechanism.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig01.png"/></fig></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Overview</title><p>This section systematically presents a 2D segmentation method for renal tumors based on CT images. First, we provide a detailed description of the construction of the experimental dataset and the standardized preprocessing pipeline. Subsequently, we comprehensively describe the design and implementation of the DeepLabV3+ framework along with its novel modules.</p></sec><sec id="s2-2"><title>Description of Experimental Data</title><p>In this study, we conducted a retrospective analysis using a private dataset obtained from the Lanzhou University Second Hospital. The dataset comprises 3D CT renal tumor images from 218 patients who were treated between 2016 and 2022. Data collection for this study was conducted over the 6-month period beginning on October 28, 2022. Notably, the dataset used in this study does not include any personal identifying information of the patients. The dataset is provided in Neuroimaging Informatics Technology Initiative (NIfTI) format with corresponding segmentation labels. In addition to CT imaging data, it encompasses various histological subtypes of renal tumors, including ccRCC, papillary renal cell carcinoma (pRCC), and chromophobe renal cell carcinoma (chRCC). The tumors were precisely classified into stages I, II, III, and IV based on the international TNM staging system. Moreover, the dataset contains fundamental clinical attributes such as patient age and gender; however, these clinical variables were collected but not used in the current segmentation algorithm, which operates solely on 2D CT images. The segmentation annotations were manually performed by experienced radiologists based on surgical pathology results to guarantee accurate tumor localization. Nontumor cystic lesions were strictly excluded to enhance the dataset&#x2019;s quality. The annotations include 2 categories: background and renal tumor, as illustrated in <xref ref-type="fig" rid="figure2">Figure 2A</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>An example cross-sectional image showing the annotations of organs and tumors in our dataset: in the private dataset, renal tumors are highlighted in orange (A), while in the public dataset, kidneys are marked in red and renal tumors in blue (B).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig02.png"/></fig><p>Additionally, experiments were conducted on the publicly available KiTS19 dataset from the 2019 Kidney and Kidney Tumor Segmentation Challenge [<xref ref-type="bibr" rid="ref33">33</xref>]. This dataset comprises original abdominal CT images and manually annotated label images by physicians, with annotation classes for background, kidney, and renal tumor, as illustrated in <xref ref-type="fig" rid="figure2">Figure 2B</xref>. To enhance data complexity and diversity, and to create a more challenging training environment for the network, we incorporated diverse cases from 210 patients. Notably, this dataset is publicly available on GitHub [<xref ref-type="bibr" rid="ref34">34</xref>] and has been approved by the Institutional Review Board of the University of Minnesota. Throughout this study, we strictly complied with the terms of the KiTS19 license [<xref ref-type="bibr" rid="ref33">33</xref>].</p><p>The private dataset and the public KiTS19 dataset exhibit complementary characteristics in terms of data construction, collectively offering diverse perspectives for evaluating model performance. The private dataset was meticulously annotated on a per-case basis by experienced radiologists, and a single representative slice was extracted for each patient using dedicated software from 3D volumetric NIfTI-format data. In contrast, the KiTS19 dataset was preprocessed using the official pipeline, generating a series of continuous slices from the raw volumetric scans. To comprehensively assess the generalizability of the proposed model, we conducted experiments on both datasets, thereby validating its robustness and adaptability across different data acquisition and annotation protocols.</p></sec><sec id="s2-3"><title>Ethical Considerations</title><p>This study was approved by the Ethics Committee of the Second Hospital of Lanzhou University (approval number: 2022A-575), ensuring compliance with medical ethical standards. Given the retrospective nature of the study and the use of deidentified data, the requirement for informed consent was waived by the ethics committee. All imaging data were fully anonymized prior to analysis, and no direct or indirect patient identifiers were accessible to the research team. No financial or material compensation was provided to participants for their inclusion in this study. No images included in the paper or supplementary materials contain identifiable information of individual participants.</p></sec><sec id="s2-4"><title>Data Preprocessing</title><sec id="s2-4-1"><title>Overview</title><p>In the data preprocessing stage, we use a variety of strategies&#x2014;including image slicing, window width and level adjustment, data augmentation, and label binarization&#x2014;to enhance both the quality of the training dataset and the efficiency of the deep learning network training. These measures enable the model to better adapt to the feature distribution of the training data. Through data augmentation, not only can more diverse and complex training samples be generated, but the model&#x2019;s generalization ability is also significantly improved, effectively preventing overfitting [<xref ref-type="bibr" rid="ref35">35</xref>]. Furthermore, to ensure consistency in data processing, all images are uniformly preprocessed to a resolution of 512&#x00D7;512 pixels for input.</p></sec><sec id="s2-4-2"><title>Windowing</title><p>Since the experimental dataset consists of 3D medical imaging data, while our segmentation network operates on a 2D architecture, it is necessary to first convert the data from 3D to 2D. For the private dataset, we used the widely used 3D Slicer software to precisely extract renal tumor slices. Meanwhile, for the KiTS19 public dataset, we used the official open-source script provided by the competition organizers to perform automated batch slicing. Only 2D CT slices containing renal tumors were retained. For the KiTS19 dataset, labels were converted to a binary schema: the kidney label (class 1) was merged into the background (class 0), retaining only background and renal tumor (class 2) categories, consistent with the annotation schema of the private dataset, while those without labeled tumors were discarded. After preprocessing, the private dataset and KiTS19 dataset yielded 218 and 4800 renal tumor images, respectively. To enhance the visibility of target tissues, we adjusted the window width (WW) and window level (WL) during the slicing process. First, we applied a dual-threshold truncation method to constrain CT values within [&#x2212;200, 500] HU, effectively reducing the impact of extreme values. This range was selected to preserve contextual anatomical information from surrounding structures, which assists in localizing the kidney and tumor relative to adjacent tissues, and is consistent with preprocessing choices in prior KiTS19 studies. Next, we determined a dynamic effective grayscale range based on the 0.5%&#x2010;99.5% quantiles and applied a Gaussian kernel filter to suppress high-frequency noise. Finally, each CT volume was standardized using <italic>z</italic>-score normalization by subtracting the volume mean and dividing by the SD. Min-max scaling was then applied to rescale the pixel values to the range [0,1] for network input. These preprocessing steps significantly improved the deep network&#x2019;s ability to extract meaningful features, as shown in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Comparison diagram for window width and window level adjustments.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig03.png"/></fig></sec><sec id="s2-4-3"><title>Data Enhancements</title><p>To mitigate overfitting and enhance the model&#x2019;s generalization ability, we used various data augmentation techniques, including flipping and rotation. By skillfully leveraging artificial expansion, we significantly enriched the diversity of the training data, thereby effectively reducing the adverse impact of data scarcity on model performance, as illustrated in <xref ref-type="fig" rid="figure4">Figure 4</xref>. During training, dynamic data augmentation is applied to each CT image using random flipping and multiangle rotations. Moreover, these transformations are randomly applied to all images with a 50% probability to ensure that the augmented data generated are sufficiently diverse and representative.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Data enhancement.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig04.png"/></fig></sec><sec id="s2-4-4"><title>Kidney Tumor Segmentation</title><p>This section provides a detailed description of the overall architecture for automatic segmentation of renal tumors in CT images. First, in the &#x201C;GAM-DeepLabV3+ Modeling&#x201D; section, we introduce the overall system design and elaborate on the implementation details of the encoder and decoder, respectively; next, in the &#x201C;MobileNetV2 Module&#x201D; section, we focus on the lightweight improvements made to the MobileNetV2 network structure; finally, in the &#x201C;GAM Module&#x201D; section, we delve into the design principles and functional mechanisms of the GAM module.</p></sec><sec id="s2-4-5"><title>GAM-DeepLabV3+ Modeling</title><p>The proposed GAM-DeepLabV3+ integrates three key components, each addressing a specific limitation of the standard DeepLabV3+: (1) the improved MobileNetV2 backbone with Efficient Channel Attention (ECA) reduces model parameters while enhancing low-level feature extraction; (2) the GAM modules in the decoder improve multiscale feature fusion and suppress background interference; and (3) the BCEDiceLoss function addresses class imbalance while optimizing boundary delineation. In recent years, DeepLabV3+ has demonstrated outstanding performance in multiscale feature extraction and fusion [<xref ref-type="bibr" rid="ref36">36</xref>]. Its encoder-decoder structure is particularly effective in segmentation tasks. However, the traditional DeepLabV3+ encoder suffers from missing small targets, while its decoder is suboptimal in capturing global information, allocating feature weights, and resisting noise, which to some extent limits segmentation accuracy. To address these limitations, we propose an improved segmentation algorithm, GAM-DeepLabV3+, the architecture of which is depicted in <xref ref-type="fig" rid="figure5">Figure 5</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>GAM-DeepLabV3+ architecture. ASPP: Atrous Spatial Pyramid Pooling; Conv: convolution; GAM: Global Attention Mechanism.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig05.png"/></fig><p>Specifically, in the encoder section, a CT image with dimensions of 512&#x00D7;512&#x00D7;3 is first processed by the backbone network for preliminary feature extraction. In contrast to the conventional use of the Xception network [<xref ref-type="bibr" rid="ref37">37</xref>], we introduce a lightweight MobileNetV2 network [<xref ref-type="bibr" rid="ref38">38</xref>] to extract both shallow and deep features. This replacement significantly reduces the computational burden while further optimizing the feature extraction process through depthwise separable convolutions, inverted residual structures, and linear bottleneck designs, which are particularly effective in preserving fine-grained information and improving the detection of small targets. Subsequently, the high-level features are fed into the Atrous Spatial Pyramid Pooling (ASPP) module [<xref ref-type="bibr" rid="ref39">39</xref>]. The ASPP module, composed of 4 convolutional layers with dilation rates of 1, 6, 12, and 18, respectively, along with a global average pooling operation, effectively integrates contextual information from different receptive fields, providing a rich and precise feature foundation for the subsequent decoder stage.</p><p>In the decoder, we introduce GAM modules at 2 critical locations (as shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>) to enhance the fusion and selection of multiscale features. First, features obtained from the encoder&#x2019;s lower layers are compressed via a 1&#x00D7;1 convolution and initially fused with high-level features; after upsampling, they are input to the first GAM module [<xref ref-type="bibr" rid="ref40">40</xref>] to highlight key information and suppress background noise. Subsequently, these attention-optimized features are merged with another set of high-level features in the second GAM module, and the results are integrated with earlier processed features via concatenation. Finally, the fused features undergo a 3&#x00D7;3 convolution and further upsampling to progressively restore resolution, ultimately producing segmentation results that match the original image dimensions. By introducing GAM modules at these 2 positions, the network is designed to better attend to edge details and to exploit global and local information at different resolutions, thereby aiming to improve segmentation accuracy and robustness while maintaining a lightweight model.</p></sec><sec id="s2-4-6"><title>MobileNetV2 Module</title><sec id="s2-4-6-1"><title>Overview</title><p>Addressing the dual requirements of network lightweight design and effective feature extraction for high-resolution medical image segmentation tasks, we adopt MobileNetV2 as the backbone encoder and optimize it accordingly. Based on depthwise separable convolutions and inverted residual blocks, MobileNetV2 implements an &#x201C;expansion&#x2013;extraction&#x2013;projection&#x201D; feature processing paradigm that balances computational complexity and feature representation, as illustrated in <xref ref-type="fig" rid="figure6">Figure 6</xref>. Initially, a 1&#x00D7;1 convolution expands the low-dimensional input into a high-dimensional feature space; then, a 3&#x00D7;3 depthwise convolution extracts spatial features; finally, a 1&#x00D7;1 convolution projects the features back to a lower dimension. In contrast to traditional residual structures with nonlinear transformations, MobileNetV2 introduces a linear bottleneck in its inverted residual blocks, which suppresses redundant nonlinear activations in high-dimensional features. This not only ensures stable gradient propagation but also provides a structural basis for preserving fine feature details.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Detailed architectures of key modules in the proposed GAM-DeepLabV3+ framework. Conv: convolution; ECA: Efficient Channel Attention; FReLU: free rectified linear unit; GAP: Global Average Pooling; ReLU: rectified linear unit.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig06.png"/></fig><p>Although MobileNetV2 exhibits efficiency in feature extraction, there remains room for improvement in the semantic discrimination of channel dimensions. To address this, we propose a hierarchical optimization strategy. In the low-level feature extraction stage, an ECA module [<xref ref-type="bibr" rid="ref41">41</xref>] is embedded after the depthwise separable convolution to enhance the expression of fundamental features such as edges and textures through adaptive channel weighting. Then, in the high-level feature processing stage, an improved free rectified linear unit (FReLU) activation function combined with a residual connection is used to optimize the gradient propagation path while preserving the integrity of the original information. This hierarchical enhancement strategy creates a complementary effect: the former calibrates channel semantics to improve local detail representation, and the latter strengthens global semantic modeling through improved nonlinear transformations. Together, they construct a hierarchical fine-grained feature expression system that maintains the network&#x2019;s lightweight advantage. The subsequent sections will discuss these 2 improvement mechanisms in detail.</p></sec><sec id="s2-4-6-2"><title>Low-Level Feature Improvement: Embedding of ECA Modules</title><p>In the inverted residual blocks of MobileNetV2, low-level features are typically extracted using 3&#x00D7;3 depthwise separable convolutions. However, this process primarily relies on local operations within each channel and lacks global modeling of interchannel relationships. To address this limitation, we introduce an ECA module between the 3&#x00D7;3 depthwise convolution (Dw Conv) and the subsequent 1&#x00D7;1 projection convolution. The purpose is to perform adaptive reweighting of channel information, thereby enhancing the network&#x2019;s ability to capture critical low-level features such as edges and textures.</p><p>The ECA module is designed to capture cross-channel dependencies through a lightweight attention mechanism, thus improving the discriminative power of feature representations. The detailed computational process of the ECA module is illustrated in <xref ref-type="fig" rid="figure6">Figure 6</xref>. For an input feature map<inline-formula><mml:math id="ieqn1"><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula>, the ECA first compresses the spatial information into a channel descriptor vector <inline-formula><mml:math id="ieqn2"><mml:mrow><mml:mi>g</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mi>C</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>via Global Average Pooling (GAP):</p><disp-formula id="E1"><label> (1)</label><mml:math id="eqn1"><mml:mrow/></mml:math></disp-formula><p>Subsequently, this descriptor vector is fed into a 1D convolution (Conv1D) over the channel dimension. The kernel size <italic>k</italic> is adaptively determined according to the channel number <italic>C</italic> as <inline-formula><mml:math id="ieqn3"><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03C8;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> , where <inline-formula><mml:math id="ieqn4"><mml:mrow><mml:mi>&#x03C8;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> denotes the nearest odd integer to <inline-formula><mml:math id="ieqn5"><mml:mrow><mml:mo>|</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mrow><mml:mi>log</mml:mi></mml:mrow><mml:mn>2</mml:mn></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>c</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>/</mml:mo><mml:mi>&#x03B3;</mml:mi><mml:mo>|</mml:mo></mml:mrow></mml:math></inline-formula>, with <italic>&#x03B3;</italic>=2 and b=1 in our implementation. This adaptive Conv1D captures local cross-channel interactions, and a Sigmoid activation function is then applied to generate channel attention weights <inline-formula><mml:math id="ieqn6"><mml:mrow><mml:mi>w</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mi>C</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula>:</p><disp-formula id="E2"><label>(2)</label><mml:math id="eqn2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>w</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x03C3;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi mathvariant="normal">C</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">v</mml:mi><mml:mn>1</mml:mn><mml:mi mathvariant="normal">D</mml:mi></mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi>g</mml:mi><mml:mo>;</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Finally, these weights are multiplied with the original feature map on a per-channel basis, yielding the enhanced feature representation:</p><disp-formula id="E3"><label>(3)</label><mml:math id="eqn3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msup><mml:mi>X</mml:mi><mml:mo>&#x2032;</mml:mo></mml:msup><mml:mo>=</mml:mo><mml:mi>w</mml:mi><mml:mo>&#x2297;</mml:mo><mml:mi>X</mml:mi></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <inline-formula><mml:math id="ieqn7"><mml:mo>&#x2297;</mml:mo></mml:math></inline-formula> denotes element-wise multiplication across channels. In this formulation, <inline-formula><mml:math id="ieqn8"><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>C</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>H</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>W</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></inline-formula> denotes the input feature map with <italic>C</italic> channels, spatial height <italic>H</italic>, and width <italic>W</italic>; <inline-formula><mml:math id="ieqn9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>g</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula> is the channel descriptor vector obtained via global average pooling; <inline-formula><mml:math id="ieqn10"><mml:mrow><mml:mi>w</mml:mi><mml:mo>&#x2208;</mml:mo><mml:msup><mml:mi>R</mml:mi><mml:mi>C</mml:mi></mml:msup></mml:mrow></mml:math></inline-formula> represents the learned channel attention weights generated by the Conv1D and Sigmoid operations; and <italic>X</italic>&#x2019; is the channel-recalibrated output feature map. This mechanism eliminates the need for an explicit fully connected layer, thereby reducing the number of parameters while maintaining computational efficiency. Moreover, the fine-grained channel reweighting effectively enhances the model&#x2019;s responsiveness to key information.</p></sec><sec id="s2-4-6-3"><title>Advanced Feature Improvement: FReLU Activation and Residual Connection</title><p>In the high-level feature extraction stage, to further enhance nonlinear expressive capability and ensure stable information propagation, we adopt an improved activation function, FReLU, and incorporate residual connections within the inverted residual blocks. Unlike the traditional rectified linear unit (ReLU) activation, which truncates negative information, FReLU uses local spatial information to generate a nonlinear mapping, thereby increasing the flexibility of feature expression. Its specific formulation can be expressed as follows:</p><disp-formula id="E4"><label>(4)</label><mml:math id="eqn4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>F</mml:mi><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>U</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03D5;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <inline-formula><mml:math id="ieqn11"><mml:mrow><mml:mi>&#x03D5;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> generated by the local convolution operation adaptively captures local feature information. In this way, not only is the positive information of the original input <italic>x</italic> preserved, but auxiliary information derived from local spatial awareness is also introduced. Furthermore, the residual connection in the inverted residual block adds the input feature <italic>X</italic> to the output <italic>F</italic>(<italic>X</italic>) obtained after a series of convolution and activation operations. This ensures stable gradient propagation and effectively preserves the initial feature information:</p><disp-formula id="E5"><label> (5)</label><mml:math id="eqn5"><mml:mrow/></mml:math></disp-formula><p>Here, <italic>X</italic> denotes the input feature map of the inverted residual block; <italic>F</italic>(<italic>X</italic>) represents the transformed feature obtained through a series of convolution, normalization, and activation operations; and <italic>H</italic>(<italic>X</italic>) denotes the residual output obtained by adding the original input <italic>X</italic> to the transformed feature <italic>F</italic>(<italic>X</italic>). This residual formulation preserves the original feature information while facilitating gradient propagation, thereby alleviating the gradient vanishing problem in deep networks. As a result, high-level features can retain richer semantic information after multiple nonlinear transformations, which is expected to support fine-grained segmentation.</p></sec><sec id="s2-4-6-4"><title>GAM Module</title><p>In renal tumor segmentation tasks, models in the decoder stage are often affected by inconsistent fusion of features from different resolutions and noise interference, which may result in misclassification of some critical features and ultimately compromise segmentation performance. To address this issue, we introduce the GAM module to enhance focus on target regions during feature fusion, as shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>. Given that both channel and spatial information are essential for refining and restoring resolution, the GAM module is incorporated at 2 critical locations within the decoder, following the 1&#x00D7;1 convolution and during the 4&#x00D7; upsampling. This placement enables the extraction of both channel and spatial features from the image.</p><p>The design inspiration for the GAM module comes from the remarkable performance of attention mechanisms in information filtering. Its structure mainly comprises 3 components: a channel attention branch, a spatial attention branch, and a residual connection, which together enable the collaborative modeling of global semantic information and local detail.</p><p>In the channel attention branch, the module first applies global average pooling to the input feature map to obtain global response information for each channel. In our implementation, the channel attention branch uses a multilayer perceptron (MLP) structure with a channel reduction ratio of 4 to model interchannel dependencies and generate importance weights for each channel. These weights reflect the contribution of each channel to the segmentation task and are multiplied element-wise with the original feature map <italic>F</italic> to obtain the channel-enhanced feature map <inline-formula><mml:math id="ieqn12"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula>, thereby reinforcing critical channel information and suppressing redundant features:</p><disp-formula id="E6"><label>(6)</label><mml:math id="eqn6"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mi>c</mml:mi></mml:msub><mml:mo>&#x2297;</mml:mo><mml:mi>F</mml:mi></mml:mrow></mml:math></disp-formula><p>Meanwhile, the spatial attention branch focuses on capturing local contextual information at each spatial location of the feature map. In our implementation, this branch consists of 2 successive 7&#x00D7;7 convolutional layers with padding of 3. The first convolution reduces the channel dimension from C to C/4, followed by batch normalization and ReLU activation, while the second convolution restores the channel dimension from C/4 to C and is followed by batch normalization. The channel reduction ratio in the spatial attention branch was set to 4. After these convolution and normalization operations, the spatial attention branch produces a spatial weight map <inline-formula><mml:math id="ieqn13"><mml:mrow><mml:msub><mml:mi>M</mml:mi><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:math></inline-formula> that matches the dimensions of the input feature map. This spatial weight map is then multiplied element-wise with <italic>F</italic><sub><italic>C</italic></sub> to obtain the spatially enhanced feature map <italic>F</italic><sub><italic>S</italic></sub>. Through dynamic adjustment of features at different spatial positions, the saliency of the target region is further enhanced:</p><disp-formula id="E7"><label>(7)</label><mml:math id="eqn7"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>F</mml:mi><mml:mi>s</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mi>s</mml:mi></mml:msub><mml:mo>&#x2297;</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mi>c</mml:mi></mml:msub></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Finally, to ensure stable propagation of features during the attention processing, the GAM module incorporates a residual connection. After the channel and spatial attention processing, the optimized features are added to the original input features, achieving shortcut information transmission. This design not only helps mitigate the gradient vanishing problem in deep networks but also preserves key information from the original features, ensuring that the introduction of the attention mechanism does not result in loss of information, thereby enhancing the overall robustness of feature representation. The schematic of this structure is shown in <xref ref-type="fig" rid="figure6">Figure 6</xref>.</p><p>Overall, by jointly applying channel and spatial attention mechanisms, the GAM module adaptively weights the input feature map, significantly enhancing focus on the target region and improving the quality of feature fusion while maintaining a lightweight model. Its flexible and efficient architecture not only boosts the model&#x2019;s capability to extract fine-grained information but also provides robust theoretical support and technical assurance for segmentation in complex scenarios.</p></sec></sec></sec><sec id="s2-5"><title>Evaluation Metrics</title><p>Fundamentally, image segmentation can be viewed as a pixel-level classification problem, where each pixel is classified as either background or target. For segmentation tasks, it is ideal for the network to correctly predict target pixels as true positives (TP) and background pixels as true negatives (TN), whereas misclassifying background pixels as target or vice versa results in false positives (FP) and false negatives (FN), respectively. In this study, the performance of the proposed GAM-DeepLabV3+ network in renal tumor segmentation is quantitatively evaluated using 6 metrics: intersection over union (IoU), Dice, recall, precision, accuracy, and 95% Hausdorff distance (HD95). IoU is a widely used metric in image segmentation that calculates the ratio of the intersection to the union of the predicted segmentation and the ground truth, as given by the following formula:</p><disp-formula id="E8"><label>(8)</label><mml:math id="eqn8"><mml:mrow><mml:mi>I</mml:mi><mml:mi>o</mml:mi><mml:mi>U</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><p>The Dice coefficient, which measures the similarity between 2 samples on a scale from 0 to 1 (with values closer to 1 indicating better performance), is calculated as follows:</p><disp-formula id="E9"><label>(9)</label><mml:math id="eqn9"><mml:mrow><mml:mi>D</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><p>Recall is defined as the proportion of TP pixels correctly identified by the model relative to all actual positive pixels in the ground truth, computed as:</p><disp-formula id="E10"><label>(10)</label><mml:math id="eqn10"><mml:mrow><mml:mi>Re</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><p>Precision refers to the ratio of TP pixels among all pixels predicted as positive by the model, computed as:</p><disp-formula id="E11"><label>(11)</label><mml:math id="eqn11"><mml:mrow><mml:mi>Pr</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><p>Accuracy denotes the proportion of all pixels that are correctly classified by the model, measuring the overall prediction accuracy with the formula:</p><disp-formula id="E12"><label> (12)</label><mml:math id="eqn12"><mml:mrow><mml:mi>A</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula><p>HD95 refers to the 95th percentile Hausdorff distance between the predicted segmentation boundary and the ground-truth boundary. It is less sensitive to extreme outliers than the maximum Hausdorff distance and was used to evaluate boundary accuracy. It is calculated as follows:</p><disp-formula id="E13"><label>(13)</label><mml:math id="eqn13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mi>H</mml:mi><mml:mi>D</mml:mi><mml:mn>95</mml:mn><mml:mo stretchy="false">(</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mo movablelimits="true" form="prefix">max</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>P</mml:mi><mml:mn>95</mml:mn><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"/></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mtext>&#x00A0;</mml:mtext><mml:mrow><mml:mo fence="true" stretchy="true" symmetric="true"/><mml:mrow><mml:mi>P</mml:mi><mml:mn>95</mml:mn><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">(</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo maxsize="1.2em" minsize="1.2em">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>where <inline-formula><mml:math id="ieqn14"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mi>d</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>B</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>a</mml:mi><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msub><mml:mo>:</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>A</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mtext>&#x00A0;</mml:mtext><mml:mtext>and</mml:mtext><mml:mtext>&#x00A0;</mml:mtext><mml:mi>d</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>B</mml:mi><mml:mo>,</mml:mo><mml:mi>A</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mo fence="false" stretchy="false">{</mml:mo><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>a</mml:mi><mml:msub><mml:mo fence="false" stretchy="false">&#x2016;</mml:mo><mml:mn>2</mml:mn></mml:msub><mml:mo>:</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x2208;</mml:mo><mml:mi>B</mml:mi><mml:mo fence="false" stretchy="false">}</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></inline-formula>. Here, A and B denote the boundary point sets of the predicted segmentation and ground-truth mask, respectively. In this study, HD95 was reported in pixels.</p><p>Our proposed loss function is a combination of binary cross-entropy (BCE) loss and Dice loss, referred to as BCEDiceLoss. BCE loss, a special case of cross-entropy loss suitable for binary classification problems, is computed as follows:</p><disp-formula id="E14"><label>(14)</label><mml:math id="eqn14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>B</mml:mi><mml:mi>C</mml:mi><mml:mi>E</mml:mi><mml:mo>=</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>y</mml:mi><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>p</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>p</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, <inline-formula><mml:math id="ieqn15"><mml:mi>y</mml:mi></mml:math></inline-formula> represents the ground truth label (0 or 1), and <inline-formula><mml:math id="ieqn16"><mml:mi>p</mml:mi></mml:math></inline-formula> denotes the model&#x2019;s predicted probability. The BCE loss function quantifies the discrepancy between the ground truth and the predicted probabilities.</p><p>By combining BCE loss with Dice loss, both classification accuracy and segmentation detail can be simultaneously optimized during training. Specifically, the total loss can be computed using the following formula:</p><disp-formula id="E15"><label>(15)</label><mml:math id="eqn15"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>B</mml:mi><mml:mi>C</mml:mi><mml:mi>E</mml:mi><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mi>B</mml:mi><mml:mi>C</mml:mi><mml:mi>E</mml:mi><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x2212;</mml:mo><mml:mi>D</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:math></disp-formula><p>Here, BCE refers to the binary cross-entropy loss, and Dice denotes the Dice coefficient (computed with a smoothing term), so that the term (1 &#x2212; Dice) constitutes the Dice loss and a higher Dice coefficient yields a lower loss. <inline-formula><mml:math id="ieqn17"><mml:mi mathvariant="normal">&#x03BB;</mml:mi></mml:math></inline-formula> is a hyperparameter used to balance the contributions of the 2 loss terms, which was empirically set to 2 in all experiments. During training, the model optimizes both BCE and Dice objectives concurrently, thereby improving its overall performance.</p><p>By integrating the metrics, we can comprehensively and objectively evaluate the model&#x2019;s performance under various conditions, thereby providing robust data support for subsequent model optimization and clinical applications.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Experimental Settings</title><p>In our experiments, we used the CT data described in the data preparation section and established the deep learning environment using Python 3.8 (Python Software Foundation) and PyTorch 1.11.0 (PyTorch Foundation, The Linux Foundation) on Ubuntu 20.04 (Canonical Ltd). The hardware configuration included an AMD EPYC 9754 processor and an RTX 4090D GPU with 24 GB of memory. During training, the batch size was set to 8 and the maximum number of epochs was set to 200, with validation performed at the end of each epoch. An early stopping strategy was used to prevent overfitting. The Adam optimizer was adopted with an initial learning rate of 1&#x00D7;10<sup>&#x2013;4</sup>, a weight decay of 1&#x00D7;10<sup>&#x2013;4</sup>, <italic>&#x03B2;</italic>1 of .9, and a minimum learning rate of 1&#x00D7;10<sup>&#x2013;5</sup>. To ensure reproducibility and assess result stability, all compared models were trained and evaluated 5 times using different random seeds (42, 123, 456, 789, and 1024), and the results are reported as mean (SD). All dataset splits were performed at the patient level to prevent data leakage. For the private dataset, 218 patients were divided into 153 training, 43 validation, and 22 testing cases, with one representative slice extracted per patient. For the KiTS19 dataset, 210 patients were divided into 147 training, 42 validation, and 21 testing cases; each patient contributed multiple continuous slices containing tumor regions, resulting in 4800 slices for training and evaluation. All backbone networks used in the ablation experiments were initialized with ImageNet pretrained weights. For the no-new-Net (nnU-Net) baseline, we adopted a 2D nnU-Net implementation and trained it under the same unified experimental protocol as the other baseline models. The complete self-configuring nnU-Net pipeline, including automated experiment planning, preprocessing, configuration selection, and postprocessing, was not used. To assess statistical significance, paired <italic>t</italic> tests were conducted across the 5 independent runs for Dice scores between the proposed method and baseline models, with statistical significance defined as <italic>P</italic>&#x003C;.001.</p></sec><sec id="s3-2"><title>Visualization Results</title><p>During the evaluation of the proposed GAM-DeepLabV3+ model, training and testing were performed on both a private dataset and the public KiTS19 dataset. All splits were performed at the patient level, with 70% for training, 20% for validation, and 10% for testing, with identical hyperparameter settings and training strategies used to objectively compare the model&#x2019;s performance across different data sources. The proposed model converged stably within 200 epochs on both datasets, with no signs of overfitting observed under the patient-level experimental setup. <xref ref-type="fig" rid="figure7">Figure 7</xref> presents the final segmentation visualization results.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Visualization of the segmentation results of the proposed model and their comparison. CT: computed tomography; FPN: feature pyramid network.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig07.png"/></fig><p>We randomly selected 2 groups of CT image predictions from the test sets of both the private dataset and the KiTS19 dataset for comparative analysis, as shown in <xref ref-type="fig" rid="figure7">Figure 7</xref>. This visual comparison demonstrates the segmentation results of different network models on renal tumors, with the white regions representing the tumor segmentation. It can be observed that traditional classic networks such as U-Net and feature pyramid network (FPN) produce relatively coarse segmentation boundaries for complex-shaped and small-volume tumor targets, often leading to segmentation errors due to insufficient deep feature extraction, which results in the failure to capture certain detailed regions. In contrast, the proposed GAM-DeepLabV3+ network performs well in segmenting small and irregular lesion areas. Compared with other classical networks, it reduces both undersegmentation and oversegmentation issues, primarily because the network is able to filter channels that contain critical information through dual attention modules, thereby reducing feature loss and enhancing overall accuracy. Consequently, it can, to a certain extent, resolve the separation between small tumor regions and the kidney, and compensate for missegmentation or omission errors observed in other networks, thereby providing valuable assistance for clinical diagnosis of renal tumors.</p></sec><sec id="s3-3"><title>Experimental Results</title><sec id="s3-3-1"><title>Overview</title><p>In our experiments, we compared the performance of the proposed network with several classical segmentation models (including U-Net, Attention-UNet, U-Net++, and ResUNet) on segmentation tasks using both a private dataset and the public KiTS19 dataset. <xref ref-type="table" rid="table1">Tables 1</xref> and <xref ref-type="table" rid="table2">2</xref> present the detailed results of each model across various metrics.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Comparison between GAM-DeepLabV3+ and other models (private dataset)<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">IoU<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup>, mean (SD)</td><td align="left" valign="bottom">Dice, mean (SD)</td><td align="left" valign="bottom">Recall, mean (SD)</td><td align="left" valign="bottom">Precision, mean (SD)</td><td align="left" valign="bottom">Accuracy, mean (SD)</td><td align="left" valign="bottom">HD95<sup><xref ref-type="table-fn" rid="table1fn3">c</xref></sup>, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">ResUNet</td><td align="left" valign="top">0.730 (0.014)</td><td align="left" valign="top">0.820 (0.008)</td><td align="left" valign="top">0.860 (0.007)</td><td align="left" valign="top">0.784 (0.009)</td><td align="left" valign="top">0.980 (0.001)</td><td align="left" valign="top">5.964 (1.731)</td></tr><tr><td align="left" valign="top">UNet</td><td align="left" valign="top">0.744 (0.017)</td><td align="left" valign="top">0.832 (0.006)</td><td align="left" valign="top">0.881 (0.008)</td><td align="left" valign="top">0.783 (0.007)</td><td align="left" valign="top">0.981 (0.002)</td><td align="left" valign="top">5.218 (1.642)</td></tr><tr><td align="left" valign="top">UNet++</td><td align="left" valign="top">0.840 (0.0017)</td><td align="left" valign="top">0.865 (0.007)</td><td align="left" valign="top">0.899 (0.010)</td><td align="left" valign="top">0.890 (0.008)</td><td align="left" valign="top">0.983 (0.001)</td><td align="left" valign="top">3.287 (1.104)</td></tr><tr><td align="left" valign="top">Attention-UNet</td><td align="left" valign="top">0.859 (0.0011)</td><td align="left" valign="top">0.867 (0.005)</td><td align="left" valign="top">0.888 (0.006)</td><td align="left" valign="top">0.893 (0.007)</td><td align="left" valign="top">0.983 (0.001)</td><td align="left" valign="top">3.012 (0.876)</td></tr><tr><td align="left" valign="top">FPN<sup><xref ref-type="table-fn" rid="table1fn4">d</xref></sup></td><td align="left" valign="top">0.866 (0.0010)</td><td align="left" valign="top">0.893 (0.008)</td><td align="left" valign="top">0.911 (0.013)</td><td align="left" valign="top">0.907 (0.005)</td><td align="left" valign="top">0.983 (0.002)</td><td align="left" valign="top">2.684 (0.932)</td></tr><tr><td align="left" valign="top">nnU-Net<sup><xref ref-type="table-fn" rid="table1fn5">e</xref></sup> (2D, custom)</td><td align="left" valign="top">0.868 (0.020)</td><td align="left" valign="top">0.908 (0.013)</td><td align="left" valign="top">0.902 (0.015)</td><td align="left" valign="top">0.891 (0.014)</td><td align="left" valign="top">0.988 (0.002)</td><td align="left" valign="top">3.936 (0.842)</td></tr><tr><td align="left" valign="top">DeepLabV3+</td><td align="left" valign="top">0.906 (0.016)</td><td align="left" valign="top">0.911 (0.008)</td><td align="left" valign="top">0.913 (0.011)</td><td align="left" valign="top">0.917 (0.008)</td><td align="left" valign="top">0.987 (0.001)</td><td align="left" valign="top">2.956 (0.642)</td></tr><tr><td align="left" valign="top">Ours</td><td align="left" valign="top">0.928 (0.012)</td><td align="left" valign="top">0.939 (0.008)</td><td align="left" valign="top">0.941 (0.007)</td><td align="left" valign="top">0.943 (0.006)</td><td align="left" valign="top">0.999 (0.000)</td><td align="left" valign="top">1.485 (0.522)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>The nnU-Net baseline in this table was implemented as a 2D nnU-Net model and trained under the same unified experimental protocol as the other baseline models, including identical patient-level data splits, optimizer settings, batch size, early stopping strategy, and random seeds. The complete self-configuring nnU-Net pipeline, including automated experiment planning, preprocessing, configuration selection, and postprocessing, was not used.</p></fn><fn id="table1fn2"><p><sup>b</sup>IoU: intersection over union.</p></fn><fn id="table1fn3"><p><sup>c</sup>HD95: 95% Hausdorff distance.</p></fn><fn id="table1fn4"><p><sup>d</sup>FPN: feature pyramid network.</p></fn><fn id="table1fn5"><p><sup>e</sup>nnU-Net: no-new-Net.</p></fn></table-wrap-foot></table-wrap><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparison between GAM-DeepLabV3+ and other models (KiTS19)<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Methods</td><td align="left" valign="bottom">IoU<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>, mean (SD)</td><td align="left" valign="bottom">Dice, mean (SD)</td><td align="left" valign="bottom">Recall, mean (SD)</td><td align="left" valign="bottom">Precision, mean (SD)</td><td align="left" valign="bottom">Accuracy, mean (SD)</td><td align="left" valign="bottom">HD95<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">ResUNet</td><td align="left" valign="top">0.872 (0.013)</td><td align="left" valign="top">0.906 (0.009)</td><td align="left" valign="top">0.881 (0.010)</td><td align="left" valign="top">0.927 (0.008)</td><td align="left" valign="top">0.987 (0.002)</td><td align="left" valign="top">3.842 (1.214)</td></tr><tr><td align="left" valign="top">FPN<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.881 (0.011)</td><td align="left" valign="top">0.912 (0.008)</td><td align="left" valign="top">0.893 (0.012)</td><td align="left" valign="top">0.931 (0.007)</td><td align="left" valign="top">0.988 (0.002)</td><td align="left" valign="top">3.516 (1.038)</td></tr><tr><td align="left" valign="top">UNet</td><td align="left" valign="top">0.888 (0.010)</td><td align="left" valign="top">0.918 (0.007)</td><td align="left" valign="top">0.905 (0.009)</td><td align="left" valign="top">0.936 (0.006)</td><td align="left" valign="top">0.986 (0.001)</td><td align="left" valign="top">3.182 (0.964)</td></tr><tr><td align="left" valign="top">Attention-UNet</td><td align="left" valign="top">0.894 (0.009)</td><td align="left" valign="top">0.921 (0.006)</td><td align="left" valign="top">0.910 (0.008)</td><td align="left" valign="top">0.938 (0.006)</td><td align="left" valign="top">0.990 (0.001)</td><td align="left" valign="top">2.964 (0.887)</td></tr><tr><td align="left" valign="top">nnU-Net<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup> (2D, custom)</td><td align="left" valign="top">0.890 (0.010)</td><td align="left" valign="top">0.915 (0.007)</td><td align="left" valign="top">0.911 (0.009)</td><td align="left" valign="top">0.934 (0.006)</td><td align="left" valign="top">0.989 (0.002)</td><td align="left" valign="top">2.970 (0.850)</td></tr><tr><td align="left" valign="top">UNet++</td><td align="left" valign="top">0.890 (0.010)</td><td align="left" valign="top">0.919 (0.007)</td><td align="left" valign="top">0.907 (0.009)</td><td align="left" valign="top">0.937 (0.006)</td><td align="left" valign="top">0.988 (0.001)</td><td align="left" valign="top">3.041 (0.912)</td></tr><tr><td align="left" valign="top">DeepLabV3+</td><td align="left" valign="top">0.897 (0.011)</td><td align="left" valign="top">0.925 (0.007)</td><td align="left" valign="top">0.910 (0.009)</td><td align="left" valign="top">0.938 (0.006)</td><td align="left" valign="top">0.991 (0.001)</td><td align="left" valign="top">2.712 (0.803)</td></tr><tr><td align="left" valign="top">Ours</td><td align="left" valign="top">0.902 (0.008)</td><td align="left" valign="top">0.928 (0.006)</td><td align="left" valign="top">0.888 (0.007)</td><td align="left" valign="top">0.939 (0.005)</td><td align="left" valign="top">0.992 (0.001</td><td align="left" valign="top">2.536 (0.742)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>The nnU-Net baseline in this table was implemented as a 2D nnU-Net model and trained under the same unified experimental protocol as the other baseline models. Because our study used 2D slice-level binary tumor segmentation, patient-level data splits, tumor-containing slice selection, and a different evaluation protocol, these results should not be directly compared with the official KiTS19 challenge leaderboard or the tumor Dice score reported by Heller et al [<xref ref-type="bibr" rid="ref33">33</xref>].</p></fn><fn id="table2fn2"><p><sup>b</sup>IoU: intersection over union.</p></fn><fn id="table2fn3"><p><sup>c</sup>HD95: 95% Hausdorff distance.</p></fn><fn id="table2fn4"><p><sup>d</sup>FPN: feature pyramid network.</p></fn><fn id="table2fn5"><p><sup>e</sup>nnU-Net: no-new-Net.</p></fn></table-wrap-foot></table-wrap><p>Note that the high accuracy value observed on the private dataset is expected because renal tumor segmentation is highly class-imbalanced, with background pixels occupying the vast majority of each CT slice. Therefore, accuracy may be inflated by correctly classified background pixels, and overlap-based metrics such as Dice and IoU provide more informative evaluation of tumor segmentation performance.</p><p>As can be seen from the data, our proposed GAM-DeepLabV3+ network achieved the best overall performance among the compared methods, particularly in terms of IoU, Dice, and HD95. Specifically, the IoU and Dice coefficients of our method on the renal tumor test set reached 92.8% and 93.9%, respectively, representing improvements of 6.2% and 4.6% over FPN. Moreover, compared with the baseline DeepLabV3+, our method achieved increases of 2.2% and 2.8% in IoU and Dice coefficients, respectively. Paired <italic>t</italic> tests based on the 5 independent runs showed that the Dice improvements of GAM-DeepLabV3+ over all compared baseline models on the private dataset were statistically significant (all <italic>P</italic>&#x003C;.001).</p><p>Similarly, as shown in <xref ref-type="table" rid="table2">Table 2</xref>, our method outperforms the other models on the public KiTS19 dataset. Paired <italic>t</italic> tests based on the 5 independent runs showed that the Dice improvements of GAM-DeepLabV3+ over all compared baseline models on the KiTS19 dataset were statistically significant (all <italic>P</italic>&#x003C;.01). On the test set, the IoU and Dice coefficients reached 90.2% and 92.8%, respectively, outperforming all compared baseline models on the KiTS19 external validation set, further demonstrating the strong generalizability of the proposed approach.</p><p>As shown in <xref ref-type="table" rid="table3">Table 3</xref>, GAM-DeepLabV3+ achieved a lower per-slice inference latency than the standard DeepLabV3+ on both the private and the public KiTS19 datasets, corresponding to a speed-up of approximately 1.5&#x00D7; on both datasets (private: from 21.8 to 14.2 ms/slice; KiTS19: from 21.8 to 14.4 ms/slice). These results highlight the efficiency and lightweight characteristics of the enhanced MobileNetV2 architecture. The experimental findings presented in the &#x201C;Experimental Results&#x201D; section systematically validate the effectiveness and practical value of the proposed method in terms of both segmentation accuracy and computational performance.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Per-slice inference latency (ms; mean, SD over 5 runs; RTX 4090D; batch size 1).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Private dataset (ms), mean (SD)</td><td align="left" valign="bottom">KiTS19 (ms), mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">DeepLabV3+</td><td align="left" valign="top">21.8 (0.4)</td><td align="left" valign="top">21.8 (0.8)</td></tr><tr><td align="left" valign="top">GAM-DeepLabV3+</td><td align="left" valign="top">14.2 (0.4)</td><td align="left" valign="top">14.4 (0.5)</td></tr></tbody></table></table-wrap><p>All results are reported as mean (SD) over 5 runs. Inference time refers to pure model forward-pass computation, excluding data loading and preprocessing.</p></sec><sec id="s3-3-2"><title>Ablation Experiment</title><p>In this section, we conducted a series of experiments to investigate the impact of different modules on the model&#x2019;s performance, mainly examining how various backbone networks and attention mechanisms integrated within the DeepLabV3+ framework affect segmentation outcomes. Our experiments aim to verify the feasibility and effectiveness of the improved MobileNetV2 network and GAM module in enhancing renal tumor segmentation performance. To this end, all backbone networks (MobileNetV2, Xception, and ResNet101) were initialized with ImageNet-pretrained weights to ensure a fair comparison and avoid performance differences attributable to training-from-scratch instability on the small private dataset. We compared the performance of each variant using quantitative metrics. The results indicate that incorporating the GAM module on top of the lightweight, improved MobileNetV2 backbone significantly enhances overall segmentation accuracy; specific data are presented in <xref ref-type="table" rid="table4">Table 4</xref>.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Ablation experiments (private dataset).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">MobileNet V2</td><td align="left" valign="bottom">Xception</td><td align="left" valign="bottom">Resnet101</td><td align="left" valign="bottom">ECA<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="bottom">CBAM<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup></td><td align="left" valign="bottom">GAM<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="bottom">IoU<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup>, mean (SD)</td><td align="left" valign="bottom">Dice, mean (SD)</td><td align="left" valign="bottom">Recall, mean (SD)</td><td align="left" valign="bottom">Precision, mean (SD)</td><td align="left" valign="bottom">Accuracy, mean (SD)</td><td align="left" valign="bottom">HD95<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup>, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">&#x2713;<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">0.910 (0.014)</td><td align="left" valign="top">0.922 (0.008)</td><td align="left" valign="top">0.925 (0.009)</td><td align="left" valign="top">0.927 (0.007)</td><td align="left" valign="top">0.987 (0.001)</td><td align="left" valign="top">2.214 (0.801)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">0.906 (0.016)</td><td align="left" valign="top">0.917 (0.008)</td><td align="left" valign="top">0.918 (0.011)</td><td align="left" valign="top">0.920 (0.008)</td><td align="left" valign="top">0.987 (0.001)</td><td align="left" valign="top">1.956 (0.642)</td></tr><tr><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">0.892 (0.013)</td><td align="left" valign="top">0.910 (0.009)</td><td align="left" valign="top">0.912 (0.010)</td><td align="left" valign="top">0.905 (0.008)</td><td align="left" valign="top">0.986 (0.001)</td><td align="left" valign="top">2.536 (0.874)</td></tr><tr><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">0.918 (0.012)</td><td align="left" valign="top">0.931 (0.008)</td><td align="left" valign="top">0.934 (0.009)</td><td align="left" valign="top">0.935 (0.007)</td><td align="left" valign="top">0.988 (0.001)</td><td align="left" valign="top">1.984 (0.732)</td></tr><tr><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top">0.914 (0.011)</td><td align="left" valign="top">0.928 (0.008)</td><td align="left" valign="top">0.930 (0.009)</td><td align="left" valign="top">0.932 (0.007)</td><td align="left" valign="top">0.988 (0.001)</td><td align="left" valign="top">2.103 (0.768)</td></tr><tr><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top"/><td align="left" valign="top">&#x2713;</td><td align="left" valign="top">0.928 (0.012)</td><td align="left" valign="top">0.939 (0.008)</td><td align="left" valign="top">0.941 (0.007)</td><td align="left" valign="top">0.943 (0.006)</td><td align="left" valign="top">0.999 (0.000)</td><td align="left" valign="top">1.485 (0.522)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>ECA: Efficient Channel Attention.</p></fn><fn id="table4fn2"><p><sup>b</sup>CBAM: Convolutional Block Attention Module.</p></fn><fn id="table4fn3"><p><sup>c</sup>GAM: Global Attention Mechanism.</p></fn><fn id="table4fn4"><p><sup>d</sup>IoU: intersection over union.</p></fn><fn id="table4fn5"><p><sup>e</sup>HD95: 95% Hausdorff distance.</p></fn><fn id="table4fn6"><p><sup>f</sup>Component included in the model configuration.</p></fn></table-wrap-foot></table-wrap><p>From the table, it can be observed that the lightweight backbone network MobileNetV2 offers advantages in model compactness and parameter efficiency while maintaining competitive accuracy. Its accuracy and robustness are slightly superior to the other backbone networks. Specifically, the IoU coefficient of MobileNetV2 is 0.4% and 1.8% higher than that of Xception and ResNet101, respectively, while its Dice coefficient exceeds those of Xception and ResNet101 by 0.5% and 1.2%, respectively. Building upon this foundation, the incorporation of the GAM further enhances model performance. After introducing GAM, the Dice coefficient and IoU improved significantly, increasing by 0.8% and 1.0%, respectively. Moreover, compared to the Convolutional Block Attention Module (CBAM) integrated at the same position, GAM achieves an additional 1.4% increase in IoU and 1.1% in the Dice coefficient. These results indicate that GAM is more effective in feature extraction and weight allocation, enabling the model to better focus on key information within the kidney tumor region. Furthermore, a comparative analysis of the evaluation metrics in the table reveals that the GAM module not only maintains a balance between precision and recall but also preserves overall segmentation accuracy. This fully validates the effectiveness and generalization capability of the GAM in kidney tumor segmentation tasks.</p></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study introduced GAM-DeepLabV3+, a lightweight, automated framework for renal tumor segmentation, and validated its efficacy on both a single-center clinical dataset and the multicenter KiTS19 benchmark. Our primary finding is that the proposed model achieves superior segmentation precision while maintaining high computational efficiency. On the private dataset, the model yielded a mean DSC of 0.939 (SD 0.008) and a mean HD95 of 1.485 (SD 0.522) pixels, consistently outperforming baseline architectures such as the standard DeepLabV3+ (DSC: mean 0.911, SD 0.008). The robust performance on the KiTS19 external validation set (DSC: mean 0.928, SD 0.006; HD95: mean 2.536, SD 0.742 pixels) further demonstrates the framework&#x2019;s strong generalizability across heterogeneous data acquisition protocols and varying annotation standards.</p><p>Ablation analysis systematically quantified the contributions of the individual architectural innovations (<xref ref-type="table" rid="table4">Table 4</xref>). Transitioning from an Xception to an enhanced MobileNetV2 backbone provided a 0.5% increase in DSC (from 0.917 to 0.922), confirming that a lightweight backbone can preserve, and even refine, feature representation while significantly reducing computational complexity. Most notably, the substitution of CBAM with the GAM resulted in the most substantial gains, improving the DSC by 1.1% over the CBAM-integrated version. This suggests that the sequential channel-spatial attention design of GAM contributes to suppressing background noise and refining boundary delineation; however, because our evaluation did not include a tumor size&#x2013;stratified analysis, a specific benefit for small renal lesions remains to be confirmed quantitatively. Furthermore, the reduction in per-slice inference latency (from 21.8 to 14.2 ms/slice on the private dataset, an approximately 1.5&#x00D7; speed-up) indicates improved computational efficiency, although real-time clinical use would require further validation.</p></sec><sec id="s4-2"><title>Comparison With Prior Work</title><p>The performance metrics in <xref ref-type="table" rid="table5">Table 5</xref> illustrate the advantages of the proposed approach relative to previously reported methods on the KiTS19 dataset. Early CNN-based approaches provided a foundation for automated renal tumor segmentation but faced inherent limitations in handling morphological complexity. Yang et al [<xref ref-type="bibr" rid="ref42">42</xref>] achieved a Dice of 82.6% using a CNN architecture with a 3-stage augmentation strategy; while this result demonstrated the promise of deep learning for this task, the approach encountered difficulties with structurally complex tumor presentations. da Cruz et al [<xref ref-type="bibr" rid="ref43">43</xref>] advanced the field by introducing a 2.5D DeepLabV3+ model with cross-plane feature fusion, reaching a Dice of 85.17%. Ji et al [<xref ref-type="bibr" rid="ref44">44</xref>] introduced ASD-Net, a U-Net variant incorporating asymmetric spatial-channel convolution modules, achieving a Dice of 85.22%. In the same period, Ruan et al [<xref ref-type="bibr" rid="ref45">45</xref>] proposed MB-FSGAN, a multiscale bipath generative adversarial network that reached 85.9%, although its feature extraction remained less sensitive to small lesions.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Comparison of segmentation metrics of the proposed method versus other state-of-the-art models (KiTS19)<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Work</td><td align="left" valign="bottom">Methods</td><td align="left" valign="bottom">Dice</td><td align="left" valign="bottom">Accuracy</td></tr></thead><tbody><tr><td align="left" valign="top">Yang et al [<xref ref-type="bibr" rid="ref42">42</xref>]</td><td align="left" valign="top">CNNs<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup></td><td align="left" valign="top">0.826</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup></td></tr><tr><td align="left" valign="top">da Cruz et al [<xref ref-type="bibr" rid="ref43">43</xref>]</td><td align="left" valign="top">DeepLabv3+2.5D</td><td align="left" valign="top">0.852</td><td align="left" valign="top">0.997</td></tr><tr><td align="left" valign="top">Ji et al [<xref ref-type="bibr" rid="ref44">44</xref>]</td><td align="left" valign="top">ASD-Net</td><td align="left" valign="top">0.852</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Ruan et al [<xref ref-type="bibr" rid="ref45">45</xref>]</td><td align="left" valign="top">MB-FSGAN</td><td align="left" valign="top">0.859</td><td align="left" valign="top">0.957</td></tr><tr><td align="left" valign="top">T&#x00FC;rk et al [<xref ref-type="bibr" rid="ref46">46</xref>]</td><td align="left" valign="top">Hybrid V-Net</td><td align="left" valign="top">0.865</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Sun et al [<xref ref-type="bibr" rid="ref47">47</xref>]</td><td align="left" valign="top">2.5D MFFAU-Net</td><td align="left" valign="top">0.875</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Jackson et al [<xref ref-type="bibr" rid="ref48">48</xref>]</td><td align="left" valign="top">3D CNNs</td><td align="left" valign="top">0.885</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">Sun et al [<xref ref-type="bibr" rid="ref49">49</xref>]</td><td align="left" valign="top">FR2PAttU-Net</td><td align="left" valign="top">0.911</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">This paper</td><td align="left" valign="top">GAM-DeepLabV3+</td><td align="left" valign="top">0.928</td><td align="left" valign="top">0.992</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>External Dice values are tumor-class scores from the original 3-class KiTS19 task (background, kidney, and tumor); the proposed method uses a binary tumor-versus-background task with the kidney merged into the background, so the comparison is indicative rather than strictly controlled.</p></fn><fn id="table5fn2"><p><sup>b</sup>CNN: convolutional neural network. </p></fn><fn id="table5fn3"><p><sup>c</sup>Component not included in the model configuration.</p></fn></table-wrap-foot></table-wrap><p>The growing interest in 3D architectures subsequently opened new directions for performance improvement. T&#x00FC;rk et al [<xref ref-type="bibr" rid="ref46">46</xref>] combined residual connections with dense block structures in a hybrid V-Net, achieving a Dice of 86.5%; however, this came at the cost of substantially increased model parameters and computational overhead. Sun et al [<xref ref-type="bibr" rid="ref47">47</xref>] balanced computational cost against accuracy through a multiscale feature fusion strategy in their 2.5D MFFAU-Net, reaching 87.5%. Jackson et al [<xref ref-type="bibr" rid="ref48">48</xref>] demonstrated that a 3D CNN augmented with data augmentation could attain a Dice of 88.5%, though the shallow architecture constrained the model&#x2019;s capacity to represent heterogeneous small-tumor features. More recently, attention mechanisms have driven further performance gains: Sun et al [<xref ref-type="bibr" rid="ref49">49</xref>] proposed FR2PAttU-Net with a dual-path feature refinement module and pyramid attention unit, achieving a Dice of 91.1%, albeit with a 32% increase in parameter count attributable to its cascaded module design. Notably, the external scores in <xref ref-type="table" rid="table5">Table 5</xref> are tumor-class Dice from the original 3-class KiTS19 task (background, kidney, and tumor), whereas our model adopts a binary tumor-versus-background formulation with the kidney merged into the background; the comparison is thus indicative rather than strictly controlled. Moreover, merging the kidney into the background does not necessarily simplify tumor delineation, as the renal parenchyma is the tissue most readily confused with tumor.</p><p>The proposed GAM-DeepLabV3+ framework achieved a Dice of 92.8% on the KiTS19 test set, which is numerically higher than the 91.1% Dice reported for FR2PAttU-Net [<xref ref-type="bibr" rid="ref49">49</xref>], while maintaining an accuracy of 99.2% and a more favorable computational profile. As noted in the footnote to <xref ref-type="table" rid="table5">Table 5</xref>, this comparison is indicative rather than a strictly controlled head-to-head result, because the external KiTS19 scores derive from the original 3-class task whereas our model adopts a binary tumor-versus-background formulation. On the private dataset, GAM-DeepLabV3+ outperformed both ASD-Net (Dice: 0.876, IoU: 0.810) and DeepLabV3+2.5D (Dice: 0.888, IoU: 0.806), achieving a mean Dice of 0.939 (SD 0.008) and a mean IoU of 0.928 (SD 0.012; <xref ref-type="table" rid="table6">Table 6</xref>). These results collectively support the effectiveness of combining global context modeling through the GAM module with a lightweight MobileNetV2 encoder for renal tumor segmentation under both data-rich and data-limited conditions. We note that these 2 comparators were reimplemented in-house and evaluated as single-run reference comparisons on the private test set, rather than under the multiseed protocol used for the other baselines in <xref ref-type="table" rid="table1">Table 1</xref>; they are therefore reported as single-point values.</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Comparison of segmentation metrics of the proposed method versus other state-of-the-art models (private dataset)<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup>.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Methods</td><td align="left" valign="bottom">IoU<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="bottom">Dice</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Accuracy</td><td align="left" valign="bottom">HD95<sup><xref ref-type="table-fn" rid="table6fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">ASD-Net</td><td align="left" valign="top">0.810</td><td align="left" valign="top">0.876</td><td align="left" valign="top">0.869</td><td align="left" valign="top">0.924</td><td align="left" valign="top">0.998</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table6fn4">d</xref></sup></td></tr><tr><td align="left" valign="top">DeepLabV3+ 2.5D</td><td align="left" valign="top">0.806</td><td align="left" valign="top">0.888</td><td align="left" valign="top">0.869</td><td align="left" valign="top">0.920</td><td align="left" valign="top">0.998</td><td align="left" valign="top">&#x2014;</td></tr><tr><td align="left" valign="top">GAM-DeepLabV3+, mean (SD)</td><td align="left" valign="top">0.928 (0.012)</td><td align="left" valign="top">0.939 (0.008)</td><td align="left" valign="top">0.941 (0.007)</td><td align="left" valign="top">0.943 (0.006)</td><td align="left" valign="top">0.999 (0.000)</td><td align="left" valign="top">1.485 (0.522)</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>ASD-Net (Ji et al [<xref ref-type="bibr" rid="ref44">44</xref>]) and DeepLabV3+ 2.5D were reimplemented by us and evaluated on the private test set as single-run reference comparisons; they were not run under the multiseed protocol used for the <xref ref-type="table" rid="table1">Table 1</xref> baselines and are therefore reported as single-point values without SDs or HD95. The ASD-Net predictions shown in <xref ref-type="fig" rid="figure7">Figure 7</xref> were produced by this reimplementation; the same reimplementation was also applied to the KiTS19 slices displayed in <xref ref-type="fig" rid="figure7">Figure 7</xref> for qualitative comparison, whereas the quantitative ASD-Net results in this table correspond to the private test set only.</p></fn><fn id="table6fn2"><p><sup>b</sup>IoU: intersection over union.</p></fn><fn id="table6fn3"><p><sup>c</sup>HD95: 95% Hausdorff distance.</p></fn><fn id="table6fn4"><p><sup>d</sup>Not available.</p></fn></table-wrap-foot></table-wrap><p>An online demonstration platform based on GAM-DeepLabV3+ has been made freely accessible at CPPDD [<xref ref-type="bibr" rid="ref50">50</xref>] (for demonstration purposes only; not intended for upload of real patient data), as illustrated in <xref ref-type="fig" rid="figure8">Figure 8</xref>. The platform is intended for research-context demonstration only and is not validated for clinical use.</p><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Homepage of the renal tumor segmentation network server.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e78523_fig08.png"/></fig></sec><sec id="s4-3"><title>Limitations</title><p>This study has several limitations that warrant consideration. The private dataset comprised 218 patients from a single center, with one representative 2D slice extracted per patient. Although this strategy is consistent with the 2D segmentation paradigm adopted here, it discards volumetric continuity information and may not fully reflect the spatial heterogeneity of renal tumors across contiguous axial slices. The relatively high performance metrics observed on the private test set may in part reflect the representative nature of the selected slices rather than the full spectrum of imaging variability encountered in clinical practice. To partially address this, external validation was performed on the KiTS19 dataset, which includes multiple consecutive slices per patient and data from multiple centers, providing a more challenging and diverse evaluation environment. The model achieved a mean HD95 of 1.485 (SD 0.522) pixels on the private test set and a mean HD95 of 2.536 (SD 0.742) pixels on KiTS19, suggesting reasonable spatial consistency of the 2D predictions; nevertheless, formal 3D reconstruction validation&#x2014;including volumetric Surface Dice and 3D HD95 computed across the full <italic>z</italic>-axis stack&#x2014;to confirm the absence of stacking artifacts remains an important direction for future investigation. This constitutes a key limitation of the current work. In addition, because the external KiTS19 results in <xref ref-type="table" rid="table5">Table 5</xref> were obtained under different task definitions (3-class vs binary), preprocessing, and evaluation protocols, the numerical differences reflect contextual competitiveness rather than a controlled demonstration of superiority. Finally, although the architecture is motivated in part by the difficulty of small renal tumors, no tumor size&#x2013;stratified evaluation was performed; the corresponding claims are therefore qualitative, and a volume-stratified analysis of small-lesion performance is left for future work. Furthermore, because only 2D slices containing annotated renal tumors were retained and tumor-free (negative) slices were discarded during dataset construction, the model&#x2019;s false-positive behavior on normal, tumor-free tissue and its performance on complete patient volumes that include negative slices were not evaluated in this study; assessment on full CT volumes therefore remains an important direction for future work.</p><p>Although nnU-Net was included as a baseline, only its 2D implementation was evaluated under our unified experimental protocol. This baseline therefore represents a custom adaptation that may underestimate the true capability of the nnU-Net framework. The complete self-configuring nnU-Net pipeline and whole-volume 3D evaluation were not explored, which should be addressed in future work. Likewise, modern ViT baselines were not included in our comparative experiments, and a direct experimental comparison against Transformer-based architectures remains future work. In addition, because of the retrospective design and limited dataset scale, the reported results should be interpreted as indicating promising accuracy for research and assistive purposes rather than as evidence of validated clinical utility. The private test set included only 22 patients, which may limit the statistical power to detect significant performance differences between methods. The heterogeneous nature of renal tumors across stages and histological subtypes, combined with the single-institution data source, may limit the generalizability of the current findings. Prospective multicenter studies with larger and more diverse patient cohorts are needed before conclusions regarding clinical deployment can be made with confidence.</p></sec><sec id="s4-4"><title>Conclusions</title><p>In summary, this study proposed GAM-DeepLabV3+, a CT renal tumor segmentation model built on an improved DeepLabV3+ framework incorporating a lightweight MobileNetV2 encoder, an ASPP module, and a GAM attention mechanism in the decoder. The model demonstrated promising segmentation accuracy on both the private dataset (Dice: mean 0.939, SD 0.008; IoU: mean 0.928, SD 0.012; HD95: mean 1.485, SD 0.522 pixels) and the public KiTS19 dataset (Dice: mean 0.928, SD 0.006; IoU: mean 0.902, SD 0.008; HD95: mean 2.536, SD 0.742 pixels), while offering improved inference speed relative to standard DeepLabV3+. By integrating lightweight backbone design with global-local attention feature fusion, the proposed approach is designed to address several challenges in renal tumor CT segmentation, such as blurred tumor boundaries and class imbalance; a dedicated, tumor size&#x2013;stratified evaluation of small-lesion performance remains an avenue for future work. These findings support the potential of GAM-DeepLabV3+ as a research tool and a basis for further development toward clinically assistive automated segmentation systems. Prospective validation on larger, multicenter datasets will be an essential next step in assessing broader clinical applicability.</p></sec></sec></body><back><ack><p>The authors used ChatGPT (OpenAI) for language editing and manuscript polishing. The authors take full responsibility for the integrity and accuracy of the content.</p></ack><notes><sec><title>Funding</title><p>This study has been supported by the Noncommunicable Chronic Diseases-National Science and Technology Major Project (Grant Number 2024ZD0524002), Gansu Provincial Natural Science Foundation (Grant Number 22JR5RA972), Jiangsu Province College Students&#x2019; Innovation and Entrepreneurship Training Program (Grant Number 202410313097Y), and the Science and Technology Program of Xuzhou (Grant Number KC25103).</p></sec><sec><title>Data Availability</title><p>The private dataset generated and analyzed during this study is not publicly available due to patient privacy restrictions, ethical requirements, and institutional data-sharing policies but may be available from the corresponding author on reasonable request, subject to institutional approval and a data-use agreement. The KiTS19 dataset analyzed during this study is available in the KiTS19 repository [<xref ref-type="bibr" rid="ref33">33</xref>]. The code generated during this study is available in the GitHub repository [<xref ref-type="bibr" rid="ref51">51</xref>]. The trained model weights are available from the corresponding author on reasonable request.</p></sec></notes><fn-group><fn fn-type="con"><p>Conceptualization: YZ, SL, JZ, BS</p><p>Data curation: YZ, JL, LS, JZ</p><p>Formal analysis: YZ</p><p>Funding acquisition: SL, BS</p><p>Investigation: YZ</p><p>Methodology: YZ</p><p>Project administration: ZL, YL, JW, XH, SL, BS</p><p>Resources: JL, SL, JZ</p><p>Software: YZ</p><p>Supervision: JL, SL, JZ, BS</p><p>Visualization: YZ</p><p>Writing &#x2013; original draft: YZ</p><p>Writing &#x2013; review &#x0026; editing: YZ, LL, ZL, YL, JW, XH, SL, JZ, BS</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">ASPP</term><def><p>Atrous Spatial Pyramid Pooling</p></def></def-item><def-item><term id="abb2">BAU-Net</term><def><p>boundary attention U-Net</p></def></def-item><def-item><term id="abb3">BCE</term><def><p>binary cross-entropy</p></def></def-item><def-item><term id="abb4">CBAM</term><def><p>Convolutional Block Attention Module</p></def></def-item><def-item><term id="abb5">ccRCC</term><def><p>clear cell renal cell carcinoma</p></def></def-item><def-item><term id="abb6">chRCC</term><def><p>chromophobe renal cell carcinoma</p></def></def-item><def-item><term id="abb7">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb8">Conv1D</term><def><p>1D convolution</p></def></def-item><def-item><term id="abb9">CRF</term><def><p>conditional random field</p></def></def-item><def-item><term id="abb10">CT</term><def><p>computed tomography</p></def></def-item><def-item><term id="abb11">DSC</term><def><p>Dice similarity coefficient</p></def></def-item><def-item><term id="abb12">Dw Conv</term><def><p>depthwise convolution</p></def></def-item><def-item><term id="abb13">ECA</term><def><p>Efficient Channel Attention</p></def></def-item><def-item><term id="abb14">FN</term><def><p>false negative</p></def></def-item><def-item><term id="abb15">FP</term><def><p>false positive</p></def></def-item><def-item><term id="abb16">FPN</term><def><p>feature pyramid network</p></def></def-item><def-item><term id="abb17">FReLU</term><def><p>free rectified linear unit</p></def></def-item><def-item><term id="abb18">GAM</term><def><p>Global Attention Mechanism</p></def></def-item><def-item><term id="abb19">GAP</term><def><p>global average pooling</p></def></def-item><def-item><term id="abb20">GPU</term><def><p>graphics processing unit</p></def></def-item><def-item><term id="abb21">HD95</term><def><p>95% Hausdorff distance</p></def></def-item><def-item><term id="abb22">IoU</term><def><p>intersection over union</p></def></def-item><def-item><term id="abb23">MLP</term><def><p>multilayer perceptron</p></def></def-item><def-item><term id="abb24">MSS U-Net</term><def><p>multiscale supervised U-Net</p></def></def-item><def-item><term id="abb25">NIfTI</term><def><p>Neuroimaging Informatics Technology Initiative</p></def></def-item><def-item><term id="abb26">nnU-Net</term><def><p>no-new-Net</p></def></def-item><def-item><term id="abb27">pRCC</term><def><p>papillary renal cell carcinoma</p></def></def-item><def-item><term id="abb28">RAU-Net</term><def><p>residual and attention U-Net</p></def></def-item><def-item><term id="abb29">ReLU</term><def><p>rectified linear unit</p></def></def-item><def-item><term id="abb30">TN</term><def><p>true negative</p></def></def-item><def-item><term id="abb31">TP</term><def><p>true positive</p></def></def-item><def-item><term id="abb32">VHL</term><def><p>von Hippel-Lindau</p></def></def-item><def-item><term id="abb33">ViT</term><def><p>Vision Transformer</p></def></def-item><def-item><term id="abb34">WHO</term><def><p>World Health Organization</p></def></def-item><def-item><term id="abb35">WL</term><def><p>window level</p></def></def-item><def-item><term id="abb36">WW</term><def><p>window width</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Baniak</surname><given-names>NJ</given-names> </name></person-group><article-title>Differential diagnosis of renal neoplasia with clear cell cytology</article-title><source>Kidney Cancer</source><year>2025</year><volume>9</volume><issue>1_suppl</issue><fpage>4</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.3233/KCA-230023</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Paner</surname><given-names>G</given-names> </name><name name-style="western"><surname>Cimadamore</surname><given-names>A</given-names> </name><name name-style="western"><surname>Franzese</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Oncocytic tumors in the kidney: a trifocal review - integrated pathological, cytopathological, and molecular perspectives (Part 1)</article-title><source>Acta Cytol</source><year>2025</year><volume>69</volume><issue>5</issue><fpage>453</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1159/000545812</pub-id><pub-id pub-id-type="medline">40245846</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vasantbhai Patel</surname><given-names>V</given-names> </name><name name-style="western"><surname>Yadav</surname><given-names>AR</given-names> </name><name name-style="western"><surname>Jain</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cenkeramaddi</surname><given-names>LR</given-names> </name></person-group><article-title>A systematic kidney tumour segmentation and classification framework using adaptive and attentive-based deep learning networks with improved crayfish optimization algorithm</article-title><source>IEEE Access</source><year>2024</year><volume>12</volume><fpage>85635</fpage><lpage>85660</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2024.3410833</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kumar</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Brar</surname><given-names>TPS</given-names> </name><name name-style="western"><surname>Kaur</surname><given-names>C</given-names> </name><name name-style="western"><surname>Singh</surname><given-names>C</given-names> </name></person-group><article-title>A comprehensive study of deep learning methods for kidney tumor, cyst, and stone diagnostics and detection using CT images</article-title><source>Arch Computat Methods Eng</source><year>2024</year><volume>31</volume><issue>7</issue><fpage>4163</fpage><lpage>4188</lpage><pub-id pub-id-type="doi">10.1007/s11831-024-10112-8</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Alaghehbandan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Siadat</surname><given-names>F</given-names> </name><name name-style="western"><surname>Trpkov</surname><given-names>K</given-names> </name></person-group><article-title>What&#x2019;s new in the WHO 2022 classification of kidney tumours?</article-title><source>Pathologica</source><year>2023</year><volume>115</volume><issue>1</issue><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.32074/1591-951X-818</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Checcucci</surname><given-names>E</given-names> </name><name name-style="western"><surname>De Cillis</surname><given-names>S</given-names> </name><name name-style="western"><surname>Granato</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Applications of neural networks in urology: a systematic review</article-title><source>Curr Opin Urol</source><year>2020</year><month>11</month><volume>30</volume><issue>6</issue><fpage>788</fpage><lpage>807</lpage><pub-id pub-id-type="doi">10.1097/MOU.0000000000000814</pub-id><pub-id pub-id-type="medline">32881726</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xie</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Prognostic value of clinical and pathological features in Chinese patients with chromophobe renal cell carcinoma: a 10-year single-center study</article-title><source>J Cancer</source><year>2017</year><volume>8</volume><issue>17</issue><fpage>3474</fpage><lpage>3479</lpage><pub-id pub-id-type="doi">10.7150/jca.19953</pub-id><pub-id pub-id-type="medline">29151931</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sadaghiani</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Baskaran</surname><given-names>S</given-names> </name><name name-style="western"><surname>Gorin</surname><given-names>MA</given-names> </name><etal/></person-group><article-title>Utility of PSMA PET/CT in staging and restaging of renal cell carcinoma: a systematic review and metaanalysis</article-title><source>J Nucl Med</source><year>2024</year><month>07</month><day>1</day><volume>65</volume><issue>7</issue><fpage>1007</fpage><lpage>1012</lpage><pub-id pub-id-type="doi">10.2967/jnumed.124.267417</pub-id><pub-id pub-id-type="medline">38782453</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Bi</surname><given-names>WL</given-names> </name><name name-style="western"><surname>Hosny</surname><given-names>A</given-names> </name><name name-style="western"><surname>Schabath</surname><given-names>MB</given-names> </name><etal/></person-group><article-title>Artificial intelligence in cancer imaging: clinical challenges and applications</article-title><source>CA Cancer J Clin</source><year>2019</year><month>03</month><volume>69</volume><issue>2</issue><fpage>127</fpage><lpage>157</lpage><pub-id pub-id-type="doi">10.3322/caac.21552</pub-id><pub-id pub-id-type="medline">30720861</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rowe</surname><given-names>SP</given-names> </name><name name-style="western"><surname>Pomper</surname><given-names>MG</given-names> </name></person-group><article-title>Molecular imaging in oncology: current impact and future directions</article-title><source>CA Cancer J Clin</source><year>2022</year><month>07</month><volume>72</volume><issue>4</issue><fpage>333</fpage><lpage>352</lpage><pub-id pub-id-type="doi">10.3322/caac.21713</pub-id><pub-id pub-id-type="medline">34902160</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>G</given-names> </name><name name-style="western"><surname>Sang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Niu</surname><given-names>Y</given-names> </name></person-group><article-title>The role of three-dimensional reconstruction in partial nephrectomy for ipsilateral multifocal renal tumors: a multicenter retrospective study</article-title><source>BMC Urol</source><year>2025</year><month>08</month><day>18</day><volume>25</volume><issue>1</issue><fpage>202</fpage><pub-id pub-id-type="doi">10.1186/s12894-025-01889-2</pub-id><pub-id pub-id-type="medline">40826402</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yu</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dai</surname><given-names>Y</given-names> </name></person-group><article-title>Crossbar-Net: a novel convolutional neural network for kidney tumor segmentation in CT images</article-title><source>IEEE Trans Image Process</source><year>2019</year><month>03</month><day>18</day><volume>28</volume><issue>8</issue><fpage>4060</fpage><lpage>4074</lpage><pub-id pub-id-type="doi">10.1109/TIP.2019.2905537</pub-id><pub-id pub-id-type="medline">30892206</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Myronenko</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hatamizadeh</surname><given-names>A</given-names> </name></person-group><article-title>3D kidneys and kidney tumor semantic segmentation using boundary-aware networks</article-title><source>arXiv</source><comment>Preprint posted online on  Sep 14, 2019</comment><pub-id pub-id-type="doi">10.48550/arXiv.1909.06684</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heller</surname><given-names>N</given-names> </name><name name-style="western"><surname>McSweeney</surname><given-names>S</given-names> </name><name name-style="western"><surname>Peterson</surname><given-names>MT</given-names> </name><etal/></person-group><article-title>An international challenge to use artificial intelligence to define the state-of-the-art in kidney and kidney tumor segmentation in CT imaging</article-title><source>J Clin Oncol</source><year>2020</year><month>02</month><day>20</day><volume>38</volume><issue>6_suppl</issue><fpage>626</fpage><pub-id pub-id-type="doi">10.1200/JCO.2020.38.6_suppl.626</pub-id><pub-id pub-id-type="medline">32706639</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>W</given-names> </name><name name-style="western"><surname>Jiang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Pe&#x00F1;a Queralta</surname><given-names>J</given-names> </name><name name-style="western"><surname>Westerlund</surname><given-names>T</given-names> </name></person-group><article-title>MSS U-Net: 3D segmentation of kidneys and tumors from CT images with a multi-scale supervised U-Net</article-title><source>Inform Med Unlocked</source><year>2020</year><volume>19</volume><fpage>100357</fpage><pub-id pub-id-type="doi">10.1016/j.imu.2020.100357</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Zeng</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Xiao</surname><given-names>J</given-names> </name></person-group><article-title>RAU-net: u-net model based on residual and attention for kidney and kidney tumor segmentation</article-title><access-date>2026-08-22</access-date><conf-name>2021 IEEE International Conference on Consumer Electronics and Computer Engineering (ICCECE)</conf-name><conf-date>Jan 15-17, 2021</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/xpl/mostRecentIssue.jsp?punumber=9341295">https://ieeexplore.ieee.org/xpl/mostRecentIssue.jsp?punumber=9341295</ext-link></comment><pub-id pub-id-type="doi">10.1109/ICCECE51280.2021.9342530</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>H</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>L</given-names> </name></person-group><article-title>Boundary attention U-Net for kidney and kidney tumor segmentation</article-title><source>Annu Int Conf Eng Med Biol Soc</source><year>2022</year><volume>2022</volume><fpage>1540</fpage><lpage>1543</lpage><pub-id pub-id-type="doi">10.1109/EMBC48229.2022.9871443</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rao</surname><given-names>PK</given-names> </name><name name-style="western"><surname>Chatterjee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Janardhan</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Optimizing inference distribution for efficient kidney tumor segmentation using a UNet-PWP deep-learning model with XAI on CT scan images</article-title><source>Diagnostics (Basel)</source><year>2023</year><month>10</month><day>18</day><volume>13</volume><issue>20</issue><fpage>3244</fpage><pub-id pub-id-type="doi">10.3390/diagnostics13203244</pub-id><pub-id pub-id-type="medline">37892065</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kele&#x015F;</surname><given-names>A</given-names> </name><name name-style="western"><surname>Crane</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Ebtag: explainable brain tumor segmentation with efficientnetv2, attention mechanism, and grad-cam in clinical and public datasets</article-title><source>SSRN</source><comment>Preprint posted online on December 3, 2024</comment><pub-id pub-id-type="doi">10.2139/ssrn.5023648</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Safarpour</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sadeghi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zarbakhsh</surname><given-names>P</given-names> </name><name name-style="western"><surname>Kamsari</surname><given-names>M</given-names> </name><name name-style="western"><surname>Kia</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name></person-group><article-title>Explainable deep learning framework for brain tumor segmentation using vision transformer and conditional random fields</article-title><source>Multimedia Systems</source><year>2026</year><month>02</month><volume>32</volume><issue>1</issue><fpage>19</fpage><pub-id pub-id-type="doi">10.1007/s00530-025-02075-y</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Choi</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Ko</surname><given-names>K</given-names> </name><name name-style="western"><surname>Baek</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>S</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>J</given-names> </name><name name-style="western"><surname>Lee</surname><given-names>M</given-names> </name></person-group><article-title>Enhanced kidney tumor segmentation in CT scans using a simplified UNETR with organ information</article-title><conf-name>2024 International Conference on Artificial Intelligence in Information and Communication (ICAIIC)</conf-name><conf-date>Feb 19-22, 2024</conf-date><pub-id pub-id-type="doi">10.1109/ICAIIC60209.2024.10463270</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdelrahman</surname><given-names>A</given-names> </name><name name-style="western"><surname>Viriri</surname><given-names>S</given-names> </name></person-group><article-title>Kidney tumor semantic segmentation using deep learning: a survey of state-of-the-art</article-title><source>J Imaging</source><year>2022</year><month>02</month><day>25</day><volume>8</volume><issue>3</issue><fpage>55</fpage><pub-id pub-id-type="doi">10.3390/jimaging8030055</pub-id><pub-id pub-id-type="medline">35324610</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Anari</surname><given-names>S</given-names> </name><name name-style="western"><surname>Crane</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bendechache</surname><given-names>M</given-names> </name></person-group><article-title>A hybrid UNet and vision transformer architecture with multi-scale fusion for brain tumor segmentation</article-title><conf-name>2024 International Conference on Medical Imaging and Computer-Aided Diagnosis (MICAD 2024)</conf-name><conf-date>Feb 19-22, 2024</conf-date><pub-id pub-id-type="doi">10.1007/978-981-96-3863-5_44</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Safarpour</surname><given-names>H</given-names> </name><name name-style="western"><surname>Anari</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Safavi</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bendechache</surname><given-names>M</given-names> </name></person-group><article-title>A dual-phase segmentation framework utilizing gumbel-softmax and a cascaded swin transformer for multi-class brain tumor segmentation</article-title><source>In Review</source><comment>Preprint posted online on July 25, 2025</comment><pub-id pub-id-type="doi">10.21203/rs.3.rs-7093467/v1</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Crane</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bendechache</surname><given-names>M</given-names> </name></person-group><article-title>The impact of backbone selection in YOLOv8 models on brain tumor localization</article-title><source>Iran J Comput Sci</source><year>2025</year><month>09</month><volume>8</volume><issue>3</issue><fpage>939</fpage><lpage>961</lpage><pub-id pub-id-type="doi">10.1007/s42044-025-00258-4</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Ranjbarzadeh</surname><given-names>R</given-names> </name><name name-style="western"><surname>Keles</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ozisik</surname><given-names>PA</given-names> </name><etal/></person-group><article-title>Combining DeepLabV3 with attention mechanisms for accurate brain tumor segmentation: insights from BraTS 2020 and a private clinical dataset</article-title><conf-name>2024 12th European Workshop on Visual Information Processing (EUVIP)</conf-name><conf-date>Sep 8-11, 2024</conf-date><pub-id pub-id-type="doi">10.1109/EUVIP61797.2024.10772776</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kan</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>GM</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>MH</given-names> </name><etal/></person-group><article-title>Automated kidney tumor segmentation in CT images using deep learning: a multi-stage approach</article-title><source>Acad Radiol</source><year>2025</year><month>12</month><volume>32</volume><issue>12</issue><fpage>7193</fpage><lpage>7203</lpage><pub-id pub-id-type="doi">10.1016/j.acra.2025.08.020</pub-id><pub-id pub-id-type="medline">40908232</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Swapna</surname><given-names>J</given-names> </name><name name-style="western"><surname>Kiruba</surname><given-names>RR</given-names> </name></person-group><article-title>Deep learning based segmentation of renal cancer detection and segmentation enhanced 3D CT images</article-title><source>J Electr Eng Technol</source><year>2026</year><month>07</month><volume>21</volume><issue>4</issue><fpage>3915</fpage><lpage>3929</lpage><pub-id pub-id-type="doi">10.1007/s42835-026-02620-3</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Automated segmentation of kidney and renal mass and automated detection of renal mass in CT urography using 3D U-Net-based deep convolutional neural network</article-title><source>Eur Radiol</source><year>2021</year><month>07</month><volume>31</volume><issue>7</issue><fpage>5021</fpage><lpage>5031</lpage><pub-id pub-id-type="doi">10.1007/s00330-020-07608-9</pub-id><pub-id pub-id-type="medline">33439313</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gido</surname><given-names>M</given-names> </name><name name-style="western"><surname>Nakagawa</surname><given-names>S</given-names> </name><name name-style="western"><surname>Mori</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kakeya</surname><given-names>H</given-names> </name></person-group><article-title>[Paper] Kidney and renal tumor segmentation by nnU-Net Using 3D CT data from different sources</article-title><source>ITE Trans Media Technol Appl</source><year>2025</year><volume>13</volume><issue>1</issue><fpage>83</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.3169/mta.13.83</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Vaswani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Shazeer</surname><given-names>N</given-names> </name><name name-style="western"><surname>Parmar</surname><given-names>N</given-names> </name><etal/></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Guyon</surname><given-names>I</given-names> </name><name name-style="western"><surname>von Luxburg</surname><given-names>U</given-names> </name><name name-style="western"><surname>Bengio</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Attention is all you need</article-title><source>Advances in Neural Information Processing Systems 30 Proceedings of the 31st Annual Conference on Neural Information Processing Systems</source><year>2017</year><publisher-name>Curran Associates</publisher-name><fpage>5998</fpage><lpage>6008</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html">https://proceedings.neurips.cc/paper_files/paper/2017/hash/3f5ee243547dee91fbd053c1c4a845aa-Abstract.html</ext-link></comment></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Karunanayake</surname><given-names>N</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>H</given-names> </name><etal/></person-group><article-title>Dual-stage AI model for enhanced CT imaging: precision segmentation of kidney and tumors</article-title><source>Tomography</source><year>2025</year><month>01</month><day>3</day><volume>11</volume><issue>1</issue><fpage>3</fpage><pub-id pub-id-type="doi">10.3390/tomography11010003</pub-id><pub-id pub-id-type="medline">39852683</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Heller</surname><given-names>N</given-names> </name><name name-style="western"><surname>Isensee</surname><given-names>F</given-names> </name><name name-style="western"><surname>Maier-Hein</surname><given-names>KH</given-names> </name><etal/></person-group><article-title>The state of the art in kidney and kidney tumor segmentation in contrast-enhanced CT imaging: results of the KiTS19 challenge</article-title><source>Med Image Anal</source><year>2021</year><month>01</month><volume>67</volume><fpage>101821</fpage><pub-id pub-id-type="doi">10.1016/j.media.2020.101821</pub-id><pub-id pub-id-type="medline">33049579</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="web"><article-title>KiTS19</article-title><source>GitHub</source><access-date>2026-08-08</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/neheller/kits19">https://github.com/neheller/kits19</ext-link></comment></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>K</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Eres-UNet++: liver CT image segmentation based on high-efficiency channel attention and Res-UNet++</article-title><source>Comput Biol Med</source><year>2023</year><month>05</month><volume>158</volume><fpage>106501</fpage><pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.106501</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Papandreou</surname><given-names>G</given-names> </name><name name-style="western"><surname>Schroff</surname><given-names>F</given-names> </name><name name-style="western"><surname>Adam</surname><given-names>H</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>Ferrari</surname><given-names>V</given-names></name><name name-style="western"><surname>Hebert</surname><given-names>M</given-names></name><name name-style="western"><surname>Sminchisescu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Weiss</surname><given-names>Y</given-names> </name></person-group><article-title>Encoder-decoder with atrous separable convolution for semantic image segmentation</article-title><source>Computer Vision &#x2013; ECCV 2018 Lecture Notes in Computer Science</source><volume>11211</volume><publisher-name>Springer</publisher-name><fpage>801</fpage><lpage>818</lpage><pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_49</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chollet</surname><given-names>F</given-names> </name></person-group><article-title>Xception: deep learning with depthwise separable convolutions</article-title><conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jul 21-26, 2017</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2017.195</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sandler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Howard</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhmoginov</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>LC</given-names> </name></person-group><article-title>MobileNetV2: inverted residuals and linear bottlenecks</article-title><conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 18-23, 2018</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>LC</given-names> </name><name name-style="western"><surname>Papandreou</surname><given-names>G</given-names> </name><name name-style="western"><surname>Kokkinos</surname><given-names>I</given-names> </name><name name-style="western"><surname>Murphy</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yuille</surname><given-names>AL</given-names> </name></person-group><article-title>DeepLab: semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected CRFs</article-title><source>IEEE Trans Pattern Anal Mach Intell</source><year>2018</year><month>04</month><volume>40</volume><issue>4</issue><fpage>834</fpage><lpage>848</lpage><pub-id pub-id-type="doi">10.1109/TPAMI.2017.2699184</pub-id><pub-id pub-id-type="medline">28463186</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Shao</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hoffmann</surname><given-names>N</given-names> </name></person-group><article-title>Global attention mechanism: retain information to enhance channel-spatial interactions</article-title><source>arXiv</source><comment>Preprint posted online on December 10 2021</comment><pub-id pub-id-type="doi">10.48550/arXiv.2112.05561</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>B</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Li</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zuo</surname><given-names>W</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Q</given-names> </name></person-group><article-title>ECA-net: efficient channel attention for deep convolutional neural networks</article-title><access-date>2026-08-23</access-date><conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 13-19, 2020</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/xpl/mostRecentIssue.jsp?punumber=9142308">https://ieeexplore.ieee.org/xpl/mostRecentIssue.jsp?punumber=9142308</ext-link></comment><pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01155</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Weakly-supervised convolutional neural networks of renal tumor segmentation in abdominal CTA images</article-title><source>BMC Med Imaging</source><year>2020</year><month>04</month><day>15</day><volume>20</volume><issue>1</issue><fpage>37</fpage><pub-id pub-id-type="doi">10.1186/s12880-020-00435-w</pub-id><pub-id pub-id-type="medline">32293303</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>da Cruz</surname><given-names>LB</given-names> </name><name name-style="western"><surname>J&#x00FA;nior</surname><given-names>DAD</given-names> </name><name name-style="western"><surname>Diniz</surname><given-names>JOB</given-names> </name><etal/></person-group><article-title>Kidney tumor segmentation from computed tomography images using DeepLabv3+ 2.5D model</article-title><source>Expert Syst Appl</source><year>2022</year><month>04</month><volume>192</volume><fpage>116270</fpage><pub-id pub-id-type="doi">10.1016/j.eswa.2021.116270</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ji</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Mu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><etal/></person-group><article-title>ASD-Net: a novel U-Net based asymmetric spatial-channel convolution network for precise kidney and kidney tumor image segmentation</article-title><source>Med Biol Eng Comput</source><year>2024</year><month>06</month><volume>62</volume><issue>6</issue><fpage>1673</fpage><lpage>1687</lpage><pub-id pub-id-type="doi">10.1007/s11517-024-03025-y</pub-id><pub-id pub-id-type="medline">38326677</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ruan</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Li</surname><given-names>D</given-names> </name><name name-style="western"><surname>Marshall</surname><given-names>H</given-names> </name><etal/></person-group><article-title>MB-FSGAN: Joint segmentation and quantification of kidney tumor on CT by the multi-branch feature sharing generative adversarial network</article-title><source>Med Image Anal</source><year>2020</year><month>08</month><volume>64</volume><fpage>101721</fpage><pub-id pub-id-type="doi">10.1016/j.media.2020.101721</pub-id><pub-id pub-id-type="medline">32554169</pub-id></nlm-citation></ref><ref id="ref46"><label>46</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>T&#x00FC;rk</surname><given-names>F</given-names> </name><name name-style="western"><surname>L&#x00FC;y</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bar&#x0131;&#x015F;&#x00E7;&#x0131;</surname><given-names>N</given-names> </name></person-group><article-title>Kidney and renal tumor segmentation using a hybrid V-Net-based model</article-title><source>Mathematics</source><year>2020</year><volume>8</volume><issue>10</issue><fpage>1772</fpage><pub-id pub-id-type="doi">10.3390/math8101772</pub-id></nlm-citation></ref><ref id="ref47"><label>47</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>F</given-names> </name><etal/></person-group><article-title>2.5D MFFAU-Net: a convolutional neural network for kidney segmentation</article-title><source>BMC Med Inform Decis Mak</source><year>2023</year><month>05</month><day>10</day><volume>23</volume><issue>1</issue><fpage>92</fpage><pub-id pub-id-type="doi">10.1186/s12911-023-02189-1</pub-id><pub-id pub-id-type="medline">37165349</pub-id></nlm-citation></ref><ref id="ref48"><label>48</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jackson</surname><given-names>P</given-names> </name><name name-style="western"><surname>Hardcastle</surname><given-names>N</given-names> </name><name name-style="western"><surname>Dawe</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kron</surname><given-names>T</given-names> </name><name name-style="western"><surname>Hofman</surname><given-names>MS</given-names> </name><name name-style="western"><surname>Hicks</surname><given-names>RJ</given-names> </name></person-group><article-title>Deep learning renal segmentation for fully automated radiation dose estimation in unsealed source therapy</article-title><source>Front Oncol</source><year>2018</year><volume>8</volume><fpage>215</fpage><pub-id pub-id-type="doi">10.3389/fonc.2018.00215</pub-id><pub-id pub-id-type="medline">29963496</pub-id></nlm-citation></ref><ref id="ref49"><label>49</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Sun</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mo</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Kidney tumor segmentation based on FR2PAttU-Net model</article-title><source>Front Oncol</source><year>2022</year><volume>12</volume><fpage>853281</fpage><pub-id pub-id-type="doi">10.3389/fonc.2022.853281</pub-id><pub-id pub-id-type="medline">35372025</pub-id></nlm-citation></ref><ref id="ref50"><label>50</label><nlm-citation citation-type="web"><article-title>KAI</article-title><source>Central Platform of Digital Disease (CPPDD)</source><access-date>2026-08-08</access-date><comment><ext-link ext-link-type="uri" xlink:href="http://www.cppdd.cn/KAI">http://www.cppdd.cn/KAI</ext-link></comment></nlm-citation></ref><ref id="ref51"><label>51</label><nlm-citation citation-type="web"><article-title>GAM-deeplabv3plus-kidneytumor</article-title><source>GitHUb</source><access-date>2026-08-08</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/yueyanzhaocs-byte/GAM-DeepLabv3Plus-KidneyTumor">https://github.com/yueyanzhaocs-byte/GAM-DeepLabv3Plus-KidneyTumor</ext-link></comment></nlm-citation></ref></ref-list></back></article>