<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e81942</article-id><article-id pub-id-type="doi">10.2196/81942</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Clinically Interpretable Deep Learning for Differentiating Vitiligo and Postinflammatory Hypopigmentation: Diagnostic Accuracy Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Alzu'bi</surname><given-names>Amal Adel</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Al Khateeb</surname><given-names>Shadi</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Elzaghmouri</surname><given-names>Bassam M</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Alshiyab</surname><given-names>Diala M</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Abu Alasal</surname><given-names>Sanaa</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff5">5</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhou</surname><given-names>Leming</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff6">6</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Computer Information Systems, Faculty of Computer &#x0026; Information Technology, Jordan University of Science and Technology</institution><addr-line>Irbid</addr-line><country>Jordan</country></aff><aff id="aff2"><institution>Department of Computer Networks, Jerash University</institution><addr-line>Jerash</addr-line><country>Jordan</country></aff><aff id="aff3"><institution>Department of Data Science and Artificial Intelligence, Al-Ahliyya Amman University</institution><addr-line>Amman</addr-line><country>Jordan</country></aff><aff id="aff4"><institution>Department of Dermatology, Jordan University of Science and Technology</institution><addr-line>Irbid</addr-line><country>Jordan</country></aff><aff id="aff5"><institution>Department of Computer Information Systems, Jordan University of Science and Technology</institution><addr-line>Irbid</addr-line><country>Jordan</country></aff><aff id="aff6"><institution>Department of Health Information Management, University of Pittsburgh</institution><addr-line>Pittsburgh</addr-line><addr-line>PA</addr-line><country>United States</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Eapen</surname><given-names>Bell</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Al-Yousef</surname><given-names>Ali</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Amal Adel Alzu'bi, PhD, Department of Computer Information Systems, Faculty of Computer &#x0026; Information Technology, Jordan University of Science and Technology, Irbid, 22110, Jordan, 962 790051705; <email>aazoubi9@just.edu.jo</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>24</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e81942</elocation-id><history><date date-type="received"><day>10</day><month>08</month><year>2025</year></date><date date-type="rev-recd"><day>15</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>16</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Amal Adel Alzu'bi, Shadi Al khateeb, Bassam M Elzaghmouri, Diala M Alshiyab, Sanaa Abu Alasal, Leming Zhou. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 24.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e81942"/><abstract><sec><title>Background</title><p>Distinguishing vitiligo from postinflammatory hypopigmentation (PIH) is clinically challenging because both conditions may present with similar depigmented lesions. Although deep learning has shown strong potential for dermatologic image classification, limited interpretability remains a barrier to clinical adoption.</p></sec><sec><title>Objective</title><p>This study aimed to develop an interpretable deep learning framework for accurate differentiation between vitiligo and PIH using a lightweight convolutional neural network and an ensemble of explainability methods.</p></sec><sec sec-type="methods"><title>Methods</title><p>A total of 332 clinical images (176 vitiligo and 156 PIH) were collected from King Abdullah University Hospital and publicly available online sources. Images were preprocessed and evaluated using patient-wise 5-fold cross-validation to eliminate patient-level data leakage. A pretrained MobileNetV2 model was fine-tuned by unfreezing the final 30 layers. To enhance interpretability, gradient-weighted class activation mapping (Grad-CAM), integrated gradients, and smooth gradients (SmoothGrad) were combined into an equal-weight ensemble explanation framework. Performance was assessed using accuracy, precision, recall, <italic>F</italic><sub>1</sub>-score, and the area under the receiver operating characteristic curve (AUC).</p></sec><sec sec-type="results"><title>Results</title><p>The proposed model achieved an overall accuracy of 94.88%, macroaveraged precision of 94.88%, recall of 94.84%, <italic>F</italic><sub>1</sub>-score of 94.86%, and an AUC of 0.9885 across the 5 validation folds. The ensemble framework produced clinically meaningful explanations in 98.48% of a representative 66-image validation subset used for interpretability analysis.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The proposed framework combines high diagnostic performance with robust interpretability for distinguishing vitiligo from PIH. By integrating multiple complementary explanation methods, the approach enhances clinical transparency and may support dermatologists in the differential diagnosis of pigmentary disorders.</p></sec></abstract><kwd-group><kwd>artificial intelligence</kwd><kwd>skin disease</kwd><kwd>vitiligo</kwd><kwd>postinflammatory hypopigmentation</kwd><kwd>deep learning</kwd><kwd>explainable AI</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Skin disorders are a leading global health issue, affecting approximately 1.9 billion people and representing the fourth leading cause of nonfatal disease burden [<xref ref-type="bibr" rid="ref1">1</xref>]. They encompass a wide range of conditions that often exhibit similar visual patterns, making timely and accurate diagnosis difficult [<xref ref-type="bibr" rid="ref2">2</xref>,<xref ref-type="bibr" rid="ref3">3</xref>]. Two such conditions, vitiligo and postinflammatory hypopigmentation (PIH), frequently present with overlapping features, namely localized depigmented patches [<xref ref-type="bibr" rid="ref4">4</xref>]. However, their etiologies differ significantly. PIH develops as a secondary reaction to skin inflammation, damage, or healing [<xref ref-type="bibr" rid="ref4">4</xref>], whereas vitiligo is an autoimmune disorder characterized by the loss of melanocytes [<xref ref-type="bibr" rid="ref5">5</xref>]. Accurately distinguishing between these disorders remains a clinical challenge, particularly in the early stages. Misdiagnosis can lead to inappropriate treatment and unnecessary psychological stress for patients.</p><p>Visual examination and clinical history play a major role in the traditional clinical diagnosis of these hypopigmented disorders [<xref ref-type="bibr" rid="ref4">4</xref>]. However, this approach is inherently subjective and may result in misguided diagnoses, particularly in resource-limited settings [<xref ref-type="bibr" rid="ref4">4</xref>]. Recent dermatological research has highlighted the diagnostic complexity of pigmentary disorders, particularly when different conditions present with similar patterns of hypopigmentation or depigmentation. Variability in lesion morphology, anatomical location, skin phototype, and disease stage can substantially affect visual appearance and complicate clinical differentiation [<xref ref-type="bibr" rid="ref6">6</xref>]. In particular, distinguishing vitiligo from PIH may be challenging because both conditions can exhibit well-demarcated pale patches with overlapping textural and pigmentary characteristics. These challenges underscore the need for objective and interpretable AI tools to support clinical decision-making and improve diagnostic consistency [<xref ref-type="bibr" rid="ref7">7</xref>]. AI-based diagnostic tools offer promising support for addressing these challenges, particularly as digital imaging and necessary computational resources for digital image processing become more widely accessible [<xref ref-type="bibr" rid="ref8">8</xref>].</p><p>Deep learning has emerged as a powerful tool in medical image analysis, particularly in dermatology, where it enables automated classification, segmentation, and anomaly detection with performance approaching that of human specialists [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref10">10</xref>]. In particular, convolutional neural networks (CNNs) have demonstrated remarkable effectiveness in skin lesion analysis, achieving dermatologist-level accuracy in various diagnostic tasks [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref11">11</xref>]. Furthermore, advanced deep learning systems have shown strong capability in the differential diagnosis of skin diseases, highlighting their potential in supporting clinical decision-making [<xref ref-type="bibr" rid="ref12">12</xref>]. Despite these advancements, the opaque nature of deep learning models, often referred to as &#x201C;black-box&#x201D; behavior, remains a significant barrier to their widespread clinical adoption, especially in sensitive conditions such as vitiligo diagnosis [<xref ref-type="bibr" rid="ref13">13</xref>].</p><p>The opaque decision-making process of CNNs raises concerns, particularly in domains such as dermatology, where interpretability is essential rather than optional [<xref ref-type="bibr" rid="ref14">14</xref>]. To address this limitation, recent studies have integrated explainable AI (XAI) methods with deep learning models [<xref ref-type="bibr" rid="ref14">14</xref>]. XAI techniques aim to improve model trustworthiness and clinical applicability by visualizing and interpreting the internal operations of neural networks [<xref ref-type="bibr" rid="ref15">15</xref>]. Gradient-weighted class activation mapping (Grad-CAM) [<xref ref-type="bibr" rid="ref16">16</xref>], integrated gradients [<xref ref-type="bibr" rid="ref17">17</xref>], and smooth gradients (SmoothGrad) [<xref ref-type="bibr" rid="ref18">18</xref>] are among the most widely used methods in this field, each providing a different perspective on model decision-making by highlighting relevant regions of input images [<xref ref-type="bibr" rid="ref16">16</xref>].</p></sec><sec id="s1-2"><title>Objective</title><p>Despite advances in XAI, individual saliency-based methods often exhibit noise, instability, and sensitivity to model architecture and parameter settings [<xref ref-type="bibr" rid="ref16">16</xref>]. To address these limitations and generate more consistent and clinically meaningful explanation maps, we propose an ensemble interpretability framework that combines Grad-CAM, integrated gradients, and SmoothGrad.</p><p>In this study, we present an interpretable deep learning framework for differentiating vitiligo and PIH using a fine-tuned MobileNetV2 architecture [<xref ref-type="bibr" rid="ref19">19</xref>]. The model was trained on dermatological images collected from both hospital and publicly available sources and evaluated using a patient-wise 5-fold cross-validation strategy. By integrating 3 complementary explainability methods, the proposed framework aims to improve the robustness of visual explanations and enhance the model&#x2019;s trustworthiness for potential use in clinical decision support systems.</p></sec><sec id="s1-3"><title>Literature Review</title><p>Deep learning has shown considerable promise in automating dermatological diagnoses, particularly for conditions such as vitiligo [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref13">13</xref>,<xref ref-type="bibr" rid="ref20">20</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. However, the clinical reliability of these models depends not only on their classification accuracy but also on their interpretability and generalizability across diverse populations and dermatological conditions.</p><p>Several studies have addressed the classification of vitiligo using deep learning, tackling challenges such as data heterogeneity, image acquisition inconsistencies, and computational efficiency. For instance, Kantoria et al [<xref ref-type="bibr" rid="ref23">23</xref>] used a dataset of 696 images and applied transfer learning with models such as Inception-V3 [<xref ref-type="bibr" rid="ref24">24</xref>], Visual Geometry Group (VGG)-16, VGG-19 [<xref ref-type="bibr" rid="ref25">25</xref>], and SqueezeNet [<xref ref-type="bibr" rid="ref26">26</xref>], followed by classifiers such as k-nearest neighbors, support vector machine, and logistic regression. Although Inception-V3 with logistic regression achieved the highest accuracy (98.0%), the method was computationally intensive, limiting its applicability in real-time settings. Similarly, Guo et al [<xref ref-type="bibr" rid="ref21">21</xref>] adopted a You Only Look Once (YOLO) v3&#x2013;based hybrid model using 2720 images for lesion classification, yet reported performance issues in detecting lesions near anatomical boundaries and challenges stemming from label noise.</p><p>To enhance image clarity, Luo et al [<xref ref-type="bibr" rid="ref27">27</xref>] proposed a Cycle-Consistent Adversarial Network (CycleGAN)&#x2013;based model incorporating super-resolution techniques, achieving a classification accuracy improvement from 76.37% to 85.69%. Despite this gain, the method&#x2019;s high computational demand posed a practical limitation for resource-limited settings. Zhang et al [<xref ref-type="bibr" rid="ref28">28</xref>] investigated multiple CNN architectures, including VGG-13, residual network (ResNet)-18, and Densely Connected Convolutional Networks (DenseNet)-121 [<xref ref-type="bibr" rid="ref29">29</xref>], across in-house and public datasets. Their models outperformed dermatologists in certain tasks; however, the datasets were demographically limited to Asian women and children, thus constraining the models&#x2019; generalizability.</p><p>Although deep learning&#x2013;based classification has been extensively explored for dermatological conditions, including hypopigmented dermatoses, no previous study has explicitly addressed the PIH using image-based techniques with visual interpretability. Several works have proposed visual explanation tools such as Grad-CAM, Grad-CAM++ [<xref ref-type="bibr" rid="ref30">30</xref>], and Score-CAM [<xref ref-type="bibr" rid="ref31">31</xref>] for skin cancer diagnosis. While these contributions have advanced the field of XAI in dermatology, none have applied these frameworks specifically to distinguish between vitiligo and PIH using clinical images.</p><p>Complementing the classification-focused studies, other research has emphasized XAI as a means to foster trust in medical diagnostics. For example, Badhon et al [<xref ref-type="bibr" rid="ref32">32</xref>] used Grad-CAM with transfer learning on a balanced dataset, achieving 97.07% accuracy while enhancing transparency through saliency maps. Zhong et al [<xref ref-type="bibr" rid="ref22">22</xref>] incorporated class activation mapping (CAM) with Swin Transformer models [<xref ref-type="bibr" rid="ref33">33</xref>], visually highlighting lesion areas, though not always aligning with clinical boundaries. Shorfuzzaman [<xref ref-type="bibr" rid="ref34">34</xref>] proposed a Shapley additive explanations (SHAP)-enhanced stacked CNN ensemble for melanoma detection, while Wang et al [<xref ref-type="bibr" rid="ref10">10</xref>] introduced a multimodal interpretability-based CNN incorporating Grad-CAM for image features and SHAP for patient metadata to improve clinical trust.</p><p>Despite these advancements, challenges remain concerning real-time feasibility, dataset diversity, and the robustness of explanation techniques. These gaps highlight the need for lightweight, interpretable, and clinically relevant deep learning workflows. Our study addresses this need by introducing a MobileNetV2-based model designed for efficient and real-time inference, coupled with an ensemble visual explanation framework that integrates Grad-CAM, integrated gradients, and SmoothGrad. To the best of our knowledge, this is the first attempt to differentiate vitiligo and PIH using explainable deep learning on clinical images, bridging a critical gap in dermatological AI.</p></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><p>In this study, we developed a robust and interpretable deep learning framework for differentiating skin images of vitiligo and PIH, 2 conditions that are clinically and visually difficult to distinguish. The proposed workflow comprised three major stages: (1) preprocessing input images to meet model requirements, including resizing, normalization, and offline augmentation; (2) training and fine-tuning a MobileNetV2-based classifier using patient-wise 5-fold cross-validation; and (3) applying 3 explainability techniques&#x2014;Grad-CAM, integrated gradients, and SmoothGrad&#x2014;to provide interpretable visual explanations of the model&#x2019;s predictions.</p><sec id="s2-1"><title>Ethical Considerations</title><p>Ethical approval was granted by the King Abdullah University Hospital (KAUH) research ethics committees (approval number 50/161/2023, date: June 8, 2023). The requirement for informed consent was waived due to the retrospective nature of the study and the use of anonymized data.</p></sec><sec id="s2-2"><title>Dataset Description and Preprocessing</title><p>The image dataset comprised 2 classes: vitiligo and PIH. A total of 332 clinical images were included, consisting of 176 vitiligo images obtained from KAUH in Irbid, Jordan, and 156 PIH images collected from KAUH and publicly accessible online sources. All images were labeled and verified by dermatologists.</p><p>The final dataset comprised 332 images obtained from 291 unique patients. Most patients contributed a single image (median=1 image per patient), while the mean number of images per patient was 1.14 (SD 0.66). The maximum number of images contributed by a single patient was 9. <xref ref-type="fig" rid="figure1">Figure 1</xref> summarizes the full distribution of images per patient. These statistics indicate that the majority of patients were represented by only 1 image, although a small number of patients contributed multiple images from different anatomical sites.</p><p>To minimize potential source-related bias arising from mixed image origins, all images were processed using a unified preprocessing pipeline, including resizing to 224&#x00D7;224 pixels, pixel-value normalization, and identical augmentation operations. Although a metadata file was used to organize image labels and patient identifiers for patient-wise cross-validation, no metadata or acquisition-related variables were provided as input features to the model. The classifier was trained exclusively on image pixel data.</p><p>The dataset was organized at the patient level, with each image linked to a unique patient identifier. Multiple images from the same patient were retained when they captured anatomically distinct lesions or body regions. During model evaluation, a patient-wise 5-fold cross-validation strategy was employed, ensuring that all images from a given patient were assigned exclusively to either the training or validation subset within each fold. This prevented patient-level data leakage and provided a more rigorous estimate of generalization to previously unseen patients.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Distribution of the number of images per patient in the final dataset.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig01.png"/></fig><p>Offline data augmentation was applied exclusively to the training subset within each fold and included random rotations, horizontal and vertical flips, brightness adjustments, and zooming operations. For each original training image, 5 augmented variants were generated, resulting in a 6-fold increase in the effective training dataset size, including the original image. The corresponding validation images were left unchanged to ensure unbiased evaluation within each fold. Because of the limited dataset size, no separate holdout or external test set was used; all performance metrics were computed exclusively from the validation folds generated during the patient-wise 5-fold cross-validation procedure.</p></sec><sec id="s2-3"><title>Model Architecture and Training Strategy</title><p>Given the need for a lightweight yet high-performing model, MobileNetV2 was selected as the backbone architecture. MobileNetV2 is a CNN optimized for mobile and embedded vision applications. The base model, pretrained on ImageNet [<xref ref-type="bibr" rid="ref35">35</xref>], was adapted using transfer learning [<xref ref-type="bibr" rid="ref36">36</xref>]. Initially, all backbone layers were frozen to leverage general visual features such as edges and textures. A subsequent fine-tuning phase was performed by unfreezing the final 30 layers of the network, specifically from block_13_expand to out_relu, enabling adaptation to dermatological features specific to vitiligo and PIH.</p><p>A custom classification head was appended to the convolutional base. This head consisted of a GlobalAveragePooling2D layer, followed by a fully connected dense layer with 128 units and ReLU activation, a dropout layer with a rate of 0.3 to reduce overfitting, and a final sigmoid-activated output layer for binary classification.</p><p>The model was trained using the Adam optimizer and binary cross-entropy loss. During the initial frozen training phase, the learning rate was set to 1&#x00D7;10&#x207B;&#x2074;. During fine-tuning, the learning rate was reduced to 1&#x00D7;10&#x207B;&#x2075; to preserve pretrained features while enabling gradual adaptation of the unfrozen layers. Training was performed for up to 25 epochs with a batch size of 32 and early stopping based on validation loss (patience=5). Hyperparameters, including learning rate, batch size, and the number of trainable layers, were selected empirically.</p><p>Model performance was evaluated using patient-wise 5-fold cross-validation. Aggregated predictions across all validation folds were used to compute the final accuracy, precision, recall, <italic>F</italic><sub>1</sub>-score, and area under the receiver operating characteristic curve (AUC). This approach provided a robust estimate of the model&#x2019;s diagnostic performance on previously unseen patients. Hyperparameters used in the model are presented in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Hyperparameters used in the MobileNet model and their values<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Hyperparameter</td><td align="left" valign="bottom">Value</td><td align="left" valign="bottom">Description</td></tr></thead><tbody><tr><td align="left" valign="top">Model architecture</td><td align="left" valign="top">MobileNetV2 (pretrained on ImageNet)</td><td align="left" valign="top">Lightweight convolutional neural network used as the backbone model</td></tr><tr><td align="left" valign="top">Input image size</td><td align="left" valign="top">224&#x00D7;224&#x00D7;3</td><td align="left" valign="top">Standard input dimensions for MobileNetV2</td></tr><tr><td align="left" valign="top">Epochs</td><td align="left" valign="top">25</td><td align="left" valign="top">Maximum number of training epochs</td></tr><tr><td align="left" valign="top">Batch size</td><td align="left" valign="top">32</td><td align="left" valign="top">Number of images processed per batch</td></tr><tr><td align="left" valign="top">Optimizer</td><td align="left" valign="top">Adam</td><td align="left" valign="top">Optimization algorithm</td></tr><tr><td align="left" valign="top">Learning rate</td><td align="left" valign="top">1&#x00D7;10&#x207B;&#x2074; (initial), 1&#x00D7;10&#x207B;&#x2075; (fine-tuning)</td><td align="left" valign="top">Learning rates used during frozen training and fine-tuning</td></tr><tr><td align="left" valign="top">Loss function</td><td align="left" valign="top">Binary cross entropy</td><td align="left" valign="top">Loss function for binary classification</td></tr><tr><td align="left" valign="top">Early stopping patience</td><td align="left" valign="top">5</td><td align="left" valign="top">Training stopped if validation loss did not improve for five consecutive epochs</td></tr><tr><td align="left" valign="top">Preprocessing</td><td align="left" valign="top">MobileNetV2 preprocess_input</td><td align="left" valign="top">Scales pixel values to the range [&#x2013;1, 1]</td></tr><tr><td align="left" valign="top">Data augmentation</td><td align="left" valign="top">Rotation, zoom, horizontal and vertical flips, brightness adjustment</td><td align="left" valign="top">Applied offline to training images only</td></tr><tr><td align="left" valign="top">Fine-tuning</td><td align="left" valign="top">Last 30 layers unfrozen</td><td align="left" valign="top">Layers from block_13_expand to out_relu were retrained</td></tr><tr><td align="left" valign="top">Dense layer units</td><td align="left" valign="top">128</td><td align="left" valign="top">Number of neurons in the fully connected layer</td></tr><tr><td align="left" valign="top">Dropout rate</td><td align="left" valign="top">0.3</td><td align="left" valign="top">Used to reduce overfitting</td></tr><tr><td align="left" valign="top">Evaluation strategy</td><td align="left" valign="top">Patient-wise 5-fold cross-validation</td><td align="left" valign="top">Prevents patient-level data leakage and provides robust performance estimates</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Name, value, and corresponding description of those hyperparameters are listed in the table.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s2-4"><title>Ensemble Explanation Framework</title><p>To enhance model interpretability, which is essential for the clinical deployment of AI systems, 3 complementary explanation methods were integrated: Grad-CAM [<xref ref-type="bibr" rid="ref16">16</xref>], integrated gradients [<xref ref-type="bibr" rid="ref17">17</xref>], and SmoothGrad [<xref ref-type="bibr" rid="ref18">18</xref>]. Each method captures a distinct perspective of feature importance within the input image.</p><list list-type="order"><list-item><p>Grad-CAM generates coarse localization maps by computing the gradients of the predicted class score with respect to the activations of the final convolutional layer, thereby highlighting class-discriminative regions.</p></list-item><list-item><p>Integrated gradients attribute the model prediction by integrating gradients along a straight-line path from a baseline image to the input image, producing pixel-level attribution scores.</p></list-item><list-item><p>SmoothGrad improves the visual quality of gradient-based saliency maps by averaging gradients computed from multiple noisy perturbations of the input image.</p></list-item></list><p>To address the limitations of relying on a single explanation technique, an ensemble explanation map was constructed by combining the outputs of the 3 methods. For each input image, the saliency maps produced by Grad-CAM, integrated gradients, and SmoothGrad were resized to a common spatial resolution using bilinear interpolation and normalized to the range [0, 1]. For integrated gradients and SmoothGrad, the absolute magnitude of the attribution values was used prior to normalization to avoid cancelation effects arising from signed gradients and to ensure consistent comparison with the nonnegative Grad-CAM activation maps.</p><p>The final ensemble map was obtained by computing the arithmetic mean of the 3 normalized saliency maps, with equal weights assigned to each method:</p><p>M_Ensemble(x) = (1/3) &#x00D7; M&#x0302;_GradCAM(x) + (1/3) &#x00D7; M&#x0302;_IG(x) + (1/3) &#x00D7; M&#x0302;_SmoothGrad(x) (1)</p><p>where x denotes the input image and M&#x0302;_GradCAM(x), M&#x0302;_IG(x), and M&#x0302;_SmoothGrad(x) represent the normalized saliency maps generated by Grad-CAM, integrated gradients, and SmoothGrad, respectively. The ensemble explanation map M_Ensemble(x) is obtained by computing the arithmetic mean of the 3 normalized maps.</p><p>This equal-weight aggregation was adopted to preserve the complementary strengths of the 3 explanation methods while reducing method-specific noise and instability. No adaptive weighting or gating mechanism was applied; therefore, all 3 methods contributed equally to the final ensemble explanation.</p></sec><sec id="s2-5"><title>Workflow Implementation</title><p>The workflow began with preprocessing the input images to satisfy the model&#x2019;s input requirements. After training and fine-tuning the MobileNetV2 classifier, Grad-CAM, integrated gradients, and SmoothGrad were applied independently to generate saliency maps for each image.</p><p>The 3 maps were subsequently normalized and averaged according to equation (1) to generate the ensemble explanation. The final output provided both the predicted class and a composite visual explanation highlighting the regions that most strongly influenced the model&#x2019;s decision, thereby improving interpretability and clinical transparency.</p></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Performance Improvement Through Fine-Tuning</title><p>Initially, the MobileNetV2 model, with all convolutional layers frozen, served as a strong baseline. <xref ref-type="fig" rid="figure2">Figure 2</xref> presents representative training and validation curves for fold 1 during the frozen training phase, including both accuracy and loss. The corresponding training and validation curves for all 5 folds are provided in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>. During this phase, the model demonstrated good discriminative ability; however, the extracted features remained largely aligned with generic ImageNet representations rather than dermatology-specific characteristics. Consequently, the model showed limitations in distinguishing subtle lesion patterns in visually ambiguous cases.</p><p>To improve domain adaptation, fine-tuning was performed by unfreezing the final 30 layers of the MobileNetV2 backbone, as described in the <italic>Methods</italic> section, specifically from the block_13_expand layer to out_relu. <xref ref-type="fig" rid="figure3">Figure 3</xref> illustrates representative training and validation curves for fold 1 during the fine-tuning phase, while the corresponding curves for all 5 folds are included in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Representative training and validation accuracy and loss curves for fold 1 during the frozen training phase, in which all MobileNetV2 backbone layers were kept nontrainable. Panel (A) shows training and validation accuracy, and panel (B) shows training and validation loss.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig02.png"/></fig><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Representative training and validation accuracy and loss curves for fold 1 during the fine-tuning phase, after unfreezing the final 30 layers of the MobileNetV2 backbone. Panel (A) shows training and validation accuracy, and panel (B) shows training and validation loss.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig03.png"/></fig><p>A temporary increase in validation loss was observed at the beginning of the fine-tuning phase. This behavior is expected because the optimizer state was reinitialized and previously frozen layers were unfrozen, substantially increasing the number of trainable parameters and altering the optimization landscape. As a result, the model required several epochs to readjust the weights before achieving stable convergence.</p><p>Following fine-tuning, the model demonstrated improved qualitative and quantitative performance. In particular, the fine-tuned model showed enhanced discrimination between depigmented vitiligo lesions and hypopigmented PIH regions, even in clinically challenging cases. The model appeared to rely on diagnostically relevant features such as lesion border sharpness, pigment uniformity, and contrast with surrounding skin.</p><p><xref ref-type="table" rid="table2">Table 2</xref> summarizes the key improvements observed after fine-tuning compared to the frozen baseline.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Comparative performance characteristics of the MobileNetV2 model before and after fine-tuning.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Metric</td><td align="left" valign="bottom">Frozen MobileNetV2</td><td align="left" valign="bottom">Fine-Tuned MobileNetV2</td></tr></thead><tbody><tr><td align="left" valign="top">Feature representation</td><td align="left" valign="top">Generic (ImageNet-based)</td><td align="left" valign="top">Domain-specific (vitiligo vs PIH<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup>)</td></tr><tr><td align="left" valign="top">Clinical differentiation</td><td align="left" valign="top">Moderate</td><td align="left" valign="top">High</td></tr><tr><td align="left" valign="top">Validation loss</td><td align="left" valign="top">Higher</td><td align="left" valign="top">Reduced</td></tr><tr><td align="left" valign="top">Generalization performance</td><td align="left" valign="top">Good</td><td align="left" valign="top">Improved</td></tr><tr><td align="left" valign="top">Overfitting risk</td><td align="left" valign="top">Low</td><td align="left" valign="top">Low (stable convergence)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>PIH: postinflammatory hypopigmentation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Model Evaluation and Performance Analysis</title><p>The final model was evaluated using patient-wise 5-fold cross-validation, ensuring that all images from a given patient were assigned exclusively to either the training or validation subset within each fold. This strategy provided a rigorous assessment of model generalization to previously unseen patients.</p><p>The number of validation images per fold ranged from 61 to 72, with no patient overlap between training and validation sets. Class balance was preserved across all folds, with each validation fold containing approximately 29&#x2010;41 vitiligo images and 30&#x2010;32 PIH images.</p><p>Across all 5 folds, the model generated 1 prediction for each of the 332 images. Aggregating these predictions yielded an overall accuracy of 94.88% and an AUC of 0.9885, indicating excellent discriminative performance. <xref ref-type="table" rid="table3">Table 3</xref> summarizes the class-wise classification metrics. Fold-specific best-validation epochs and stopping epochs for both the frozen training and fine-tuning phases are provided in Table S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>These results demonstrate strong and well-balanced diagnostic performance across both classes. The comparable precision and recall values indicate that the model does not exhibit a substantial bias toward either vitiligo or PIH.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Classification performance of the fine-tuned MobileNetV2 model using patient-wise 5-fold cross-validation<sup><xref ref-type="table-fn" rid="table3fn1">a,b</xref></sup>.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td></tr></thead><tbody><tr><td align="left" valign="top">PIH<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td><td align="left" valign="top">0.9484</td><td align="left" valign="top">0.9423</td><td align="left" valign="top">0.9453</td></tr><tr><td align="left" valign="top">Vitiligo</td><td align="left" valign="top">0.9492</td><td align="left" valign="top">0.9545</td><td align="left" valign="top">0.9518</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>Overall accuracy: 0.9488.</p></fn><fn id="table3fn2"><p><sup>b</sup>Area under the receiver operating characteristic curve: 0.9885.</p></fn><fn id="table3fn3"><p><sup>c</sup>PIH: postinflammatory hypopigmentation.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Classification Metrics and Diagnostic Performance</title><p>The aggregated confusion matrix (<xref ref-type="fig" rid="figure4">Figure 4</xref>) summarizes the predictions across all 5 validation folds, with each image contributing exactly once to the final evaluation.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Aggregated confusion matrix of the fine-tuned MobileNetV2 model for differentiating vitiligo from postinflammatory hypopigmentation (PIH). The matrix includes predictions from all 5 validation folds, with each image appearing exactly once.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig04.png"/></fig><p>The model correctly classified 147 of 156 PIH images and 168 of 176 vitiligo images, resulting in only 17 misclassifications across the entire dataset. Specifically, 9 PIH images were incorrectly classified as vitiligo, whereas 8 vitiligo images were incorrectly classified as PIH.</p><p>The relatively small and nearly symmetric number of false-positive and false-negative predictions further confirms the balanced classification behavior of the model and supports its potential use for differentiating these visually similar pigmentary disorders.</p><p>The receiver operating characteristic (ROC) curve aggregated across all 5 validation folds (<xref ref-type="fig" rid="figure5">Figure 5</xref>) yielded an AUC of 0.9885, demonstrating excellent separability between vitiligo and PIH.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Aggregated receiver operating characteristic (ROC) curve of the fine-tuned MobileNetV2 model across all 5 validation folds. The dashed diagonal line represents random classification performance (area under the receiver operating characteristic curve [AUC]=0.50).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig05.png"/></fig></sec><sec id="s3-4"><title>Model Interpretability Through Ensemble Explanation Methods</title><sec id="s3-4-1"><title>Overview</title><p>Interpretability analysis was performed using Grad-CAM, integrated gradients, SmoothGrad, and the proposed ensemble explanation framework. Four recurring interpretability scenarios were identified based on the degree of agreement and complementarity among the explanation methods. Representative examples are presented in <xref ref-type="fig" rid="figure6">Figures 6</xref><xref ref-type="fig" rid="figure7"/><xref ref-type="fig" rid="figure8"/>-<xref ref-type="fig" rid="figure9">9</xref>.</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Representative example of scenario 1, where gradient-weighted class activation mapping (Grad-CAM), integrated gradients, smooth gradients (SmoothGrad), and the ensemble map consistently highlighted the same lesion region, demonstrating strong agreement across all explanation methods.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig06.png"/></fig><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Representative example of scenario 2, where 1 explanation method produced a suboptimal attribution map, whereas the ensemble map remained robust and accurately localized the lesion. Grad-CAM: gradient-weighted class activation mapping; SmoothGrad: smooth gradients.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig07.png"/></fig><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Representative example of scenario 3, where the individual explanation methods emphasized complementary lesion characteristics and the ensemble map integrated them into a unified interpretation. Grad-CAM: gradient-weighted class activation mapping; SmoothGrad: smooth gradients.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig08.png"/></fig><fig position="float" id="figure9"><label>Figure 9.</label><caption><p>Representative example of scenario 4, where all explanation methods produced noisy or partially inconsistent saliency maps, resulting in a less informative ensemble map. Grad-CAM: gradient-weighted class activation mapping; SmoothGrad: smooth gradients.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e81942_fig09.png"/></fig></sec><sec id="s3-4-2"><title>Scenario 1: High Agreement Across All Methods</title><p>In the most common scenario, all 3 explanation methods generated highly consistent saliency maps, with each method focusing on the same clinically relevant lesion region. The resulting ensemble map (<xref ref-type="fig" rid="figure6">Figure 6</xref>) closely matched the individual explanations and reinforced the confidence in the model&#x2019;s prediction.</p><p>This scenario was observed in 36 of 66 (54.55%) validation images, indicating that the 3 methods frequently converged on similar and clinically meaningful attribution patterns.</p></sec><sec id="s3-4-3"><title>Scenario 2: Failure of a Single Explanation Method</title><p>In this scenario (<xref ref-type="fig" rid="figure7">Figure 7</xref>), 1 explanation method produced a weak, noisy, or uninformative saliency map, while the other 2 methods successfully localized the lesion. Despite the failure of 1 individual method, the ensemble map remained accurate and clinically meaningful by integrating the informative signals from the remaining methods.</p><p>This pattern was observed in 25 of 66 (37.88%) images, demonstrating that the ensemble framework effectively mitigated isolated failures of individual explanation techniques.</p></sec><sec id="s3-4-4"><title>Scenario 3: Complementary Information Across Methods</title><p>In a smaller subset of cases, the 3 explanation methods highlighted different but complementary aspects of the lesion. For example, 1 method emphasized lesion borders, whereas another focused on internal pigmentation changes. By combining these complementary signals, the ensemble map provided a more comprehensive and informative explanation than any single method alone (<xref ref-type="fig" rid="figure8">Figure 8</xref>). This scenario was identified in 4 of 66 (6.06%) images.</p></sec><sec id="s3-4-5"><title>Scenario 4: All Methods Are Noisy or Partially Inconsistent</title><p>In 1 challenging case, all 3 explanation methods produced noisy or partially inconsistent saliency maps. The ensemble map remained similarly diffuse, providing limited improvement over the individual methods. <xref ref-type="fig" rid="figure9">Figure 9</xref> illustrates this case.</p><p>This scenario occurred in only 1 of 66 (1.52%) images, indicating that the proposed framework generated clinically useful explanations in the vast majority of cases.</p></sec></sec><sec id="s3-5"><title>Quantitative Distribution of Interpretability Scenarios</title><p>To quantitatively assess the prevalence of these interpretability patterns beyond the representative examples shown in <xref ref-type="fig" rid="figure6">Figures 6</xref><xref ref-type="fig" rid="figure7"/><xref ref-type="fig" rid="figure8"/>-<xref ref-type="fig" rid="figure9">9</xref>, <xref ref-type="table" rid="table4">Table 4</xref> summarizes the distribution of images across the 4 scenarios within a complete validation fold containing 66 images.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Distribution of validation images across 4 interpretability scenarios (n=66).</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Scenario</td><td align="left" valign="bottom">Description</td><td align="left" valign="bottom">Images, n (%)</td></tr></thead><tbody><tr><td align="left" valign="top">1</td><td align="left" valign="top">High agreement across all methods</td><td align="left" valign="top">36 (54.55)</td></tr><tr><td align="left" valign="top">2</td><td align="left" valign="top">One method fails, but the ensemble remains robust</td><td align="left" valign="top">25 (37.88)</td></tr><tr><td align="left" valign="top">3</td><td align="left" valign="top">Methods provide complementary information</td><td align="left" valign="top">4 (6.06)</td></tr><tr><td align="left" valign="top">4</td><td align="left" valign="top">All methods are noisy or partially inconsistent</td><td align="left" valign="top">1 (1.52)</td></tr><tr><td align="left" valign="top">Total</td><td align="left" valign="top">&#x2014;<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="top">66 (100)</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>Not applicable.</p></fn></table-wrap-foot></table-wrap><p>A single validation fold was selected as a representative subset of the dataset. This approach is consistent with established practices in explainable artificial intelligence, where qualitative and case-based interpretation analyses are commonly conducted on representative samples rather than the entire dataset because of computational costs and the inherently qualitative nature of visual assessment [<xref ref-type="bibr" rid="ref37">37</xref>]. Furthermore, within the 5-fold cross-validation framework, each validation fold constitutes a statistically representative partition of the overall dataset, preserving class balance and lesion variability [<xref ref-type="bibr" rid="ref16">16</xref>].</p><p>As shown in <xref ref-type="table" rid="table4">Table 4</xref>, the ensemble framework produced clinically meaningful explanations in 65 (98.48%) of 66 images, corresponding to scenarios 1&#x2010;3. Scenario 1, representing strong agreement across all methods, was the most frequent outcome. Scenario 2 was also common, demonstrating that the ensemble remained reliable even when 1 individual method failed. Scenario 4 was rare, occurring in only a single image.</p></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>This study developed and validated an interpretable deep learning framework for differentiating vitiligo from PIH, 2 pigmentary disorders that are often difficult to distinguish because of their overlapping clinical appearance. Using a fine-tuned MobileNetV2 model evaluated with patient-wise 5-fold cross-validation, the proposed framework achieved an overall accuracy of 94.88%, a macroaveraged precision of 94.88%, a recall of 94.84%, and an <italic>F</italic><sub>1</sub>-score of 94.86%, demonstrating balanced classification performance across both classes, and an AUC of 0.9885.</p><p>Although patient-wise partitioning generally provides a more conservative estimate of performance, the revised pipeline also involved the reconstruction of the full preprocessing, metadata organization, fold generation, and model evaluation workflow. Because the original dataset already contained minimal patient overlap (mean 1.14, SD 0.66 images per patient; median 1, IQR 1-1, range 1-9 images per patient), patient-level leakage in the earlier image-wise protocol was likely limited. The observed performance difference, therefore, most likely reflects a combination of revised fold composition, random initialization effects, and improved stability of the rebuilt evaluation pipeline rather than leakage alone.</p><p>These results indicate that the model achieved high and well-balanced diagnostic performance for both classes. The aggregated confusion matrix demonstrated that 315 of 332 images were correctly classified, with only 17 misclassifications across all validation folds. This performance is particularly notable given that the evaluation protocol ensured that images from the same patient were never shared between training and validation subsets, thereby eliminating patient-level data leakage and providing a more rigorous estimate of generalization to previously unseen patients.</p><p>In addition to strong quantitative performance, the proposed framework incorporated an ensemble explainability strategy that combined Grad-CAM, integrated gradients, and SmoothGrad. The resulting explanation maps consistently highlighted clinically relevant features, including lesion border sharpness, pigment uniformity, and contrast with surrounding skin, indicating that the model based its predictions on diagnostically meaningful visual patterns rather than background artifacts or acquisition-specific cues.</p><p>From a clinical perspective, differentiating vitiligo from PIH is inherently challenging, particularly in early-stage lesions or in cases with subtle pigmentary changes. The strong performance achieved under a strict patient-wise validation strategy suggests that the model learned robust lesion-specific features that generalize beyond individual patients and image sources.</p></sec><sec id="s4-2"><title>Impact of Fine-Tuning and Domain Adaptation</title><p>The initial frozen training phase provided a strong baseline by leveraging generic visual features learned from ImageNet pretraining. However, these features are not optimized to capture the subtle morphological and pigmentary characteristics that distinguish vitiligo from PIH.</p><p>Fine-tuning the final 30 layers of MobileNetV2 enabled the network to adapt to dermatology-specific patterns. This adaptation improved both quantitative performance and qualitative localization, as evidenced by the stable convergence behavior observed in the training curves and the enhanced focus on lesion-relevant regions in the saliency maps.</p><p>A transient increase in validation loss was observed at the beginning of the fine-tuning phase. This behavior is expected when previously frozen layers are unfrozen and the optimizer state is reinitialized, resulting in a temporary adjustment period before the model converges to a new optimum.</p></sec><sec id="s4-3"><title>Cross-Validation Stability and Generalization</title><p>The patient-wise 5-fold cross-validation framework demonstrated consistent performance across all folds, supporting the robustness of the proposed approach. Because each validation fold contained images from entirely unseen patients, the resulting metrics provide a conservative and clinically realistic estimate of diagnostic performance.</p><p>The use of aggregated predictions across all validation folds further strengthened the analysis by enabling computation of final performance metrics from the complete set of 332 out-of-sample predictions rather than from a single fold. This approach reduced the influence of fold-specific variability and yielded a comprehensive assessment of model behavior.</p></sec><sec id="s4-4"><title>Error Analysis</title><p>Although the overall performance was highly balanced, a small number of errors remained. Specifically, 9 PIH images were misclassified as vitiligo, and 8 vitiligo images were misclassified as PIH.</p><p>These errors likely reflect the substantial clinical overlap between the 2 conditions. Early vitiligo lesions may exhibit partial depigmentation or indistinct borders, whereas PIH lesions may present with sharply demarcated hypopigmentation that closely resembles vitiligo. Such borderline cases can be challenging even for experienced dermatologists.</p><p>Despite these difficult cases, the low number of misclassifications and the high AUC of 0.9885 indicate excellent discrimination across a wide range of decision thresholds.</p></sec><sec id="s4-5"><title>Interpretability Analysis and Value of the Ensemble Framework</title><p>Interpretability analysis demonstrated that the ensemble explanation framework substantially improved the robustness of visual explanations.</p><p>A complete validation fold containing 66 images was manually reviewed and categorized into 4 predefined interpretability scenarios. In 36 (54.55%) images, all 3 explanation methods showed strong agreement (scenario 1). In 25 (37.88%) images, 1 method produced a weak or noisy map, but the ensemble remained robust by integrating informative signals from the remaining methods (scenario 2). In 4 (6.06%) images, the methods provided complementary information that enriched the final explanation (scenario 3). Only 1 (1.52%) image exhibited noisy or partially inconsistent maps across all methods (scenario 4).</p><p>Overall, the ensemble framework produced clinically meaningful explanations in 65 (98.48%) of 66 images. These findings confirm that combining multiple explanation techniques provides greater stability and reliability than relying on any single method alone.</p></sec><sec id="s4-6"><title>Comparison With Prior Work</title><p>While explainability techniques such as Grad-CAM have been extensively explored in dermatological AI applications, particularly for skin cancer detection [<xref ref-type="bibr" rid="ref10">10</xref>,<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref32">32</xref>], their application to hypopigmented disorders remains limited. Previous studies have demonstrated the value of visual explanations for building clinical trust. For example, Badhon et al [<xref ref-type="bibr" rid="ref32">32</xref>] and Shorfuzzaman [<xref ref-type="bibr" rid="ref34">34</xref>] demonstrated increased trust via saliency maps but rarely benchmarked multiple explanation methods together or addressed conditions such as PIH. These studies typically relied on single explainability methods and did not systematically address their individual limitations or explore multimethod ensemble approaches. In contrast, this study integrates 3 complementary explanation methods into a unified ensemble framework and quantitatively evaluates their robustness through predefined interpretability scenarios.</p><p>The use of patient-wise cross-validation and the high diagnostic performance achieved under this stricter protocol further strengthen the clinical relevance of the proposed approach.</p></sec><sec id="s4-7"><title>Clinical Implications</title><p>The combination of high classification performance and robust interpretability makes the proposed framework particularly attractive for clinical decision support.</p><p>The lightweight MobileNetV2 architecture offers computational efficiency suitable for deployment in resource-limited environments, mobile apps, and teledermatology platforms. At the same time, the ensemble explanation maps provide transparent visual evidence that can help dermatologists understand and verify the basis of model predictions.</p><p>Rather than replacing clinician judgment, the framework is intended to serve as an assistive tool that supports more consistent and objective differentiation between vitiligo and PIH.</p></sec><sec id="s4-8"><title>Limitations</title><p>Several limitations should be acknowledged.</p><p>First, although the dataset size was sufficient to demonstrate proof of concept, it remains relatively modest for deep learning applications. Larger multicenter datasets would be valuable for further validation.</p><p>Second, most images were obtained from a single institution, supplemented by publicly available web images. Although all images were processed using a unified preprocessing pipeline, additional validation on external cohorts would strengthen generalizability.</p><p>Third, systematic subgroup analyses based on Fitzpatrick skin type, age, ethnicity, lesion location, and disease duration were not possible because these metadata were not consistently available.</p><p>Fourth, the model relied exclusively on image data and did not incorporate clinical information such as lesion history, preceding inflammation, or family history, which may further improve diagnostic performance.</p><p>Finally, while the explainability analysis was conducted on a representative validation fold, evaluation across all folds and formal assessment by dermatologists would provide additional evidence regarding explanation quality and clinical usefulness.</p></sec><sec id="s4-9"><title>Future Directions</title><p>Several directions may extend the present work. First, multicenter studies involving larger and more diverse datasets are needed to validate the model across different imaging conditions, patient populations, and skin phototypes. Collecting structured demographic and clinical variables, such as Fitzpatrick skin type, age, ethnicity, lesion location, and disease duration, would enable subgroup analyses to assess fairness and generalizability.</p><p>Second, incorporating structured clinical metadata, including lesion onset, history of preceding inflammation, and family history, may improve diagnostic performance and better reflect real-world dermatological decision-making.</p><p>Third, the proposed ensemble explainability framework could be further refined by investigating adaptive weighting strategies and quantitatively validated against expert-annotated lesion boundaries and dermatologist assessments.</p><p>Fourth, future studies should compare MobileNetV2 with alternative architectures, including larger convolutional models such as ResNet and EfficientNet [<xref ref-type="bibr" rid="ref38">38</xref>] and recent vision foundation models [<xref ref-type="bibr" rid="ref39">39</xref>,<xref ref-type="bibr" rid="ref40">40</xref>], to evaluate the trade-off between predictive performance and computational efficiency.</p><p>Finally, prospective clinical studies are needed to assess the impact of the proposed framework on diagnostic accuracy, clinician confidence, and workflow efficiency, particularly in teledermatology and resource-limited settings.</p></sec><sec id="s4-10"><title>Conclusions</title><p>This study developed an interpretable deep learning framework for differentiating vitiligo from PIH using a fine-tuned MobileNetV2 model and an ensemble of Grad-CAM, integrated gradients, and SmoothGrad.</p><p>Under a rigorous patient-wise 5-fold cross-validation protocol, the model achieved excellent diagnostic performance, with an overall accuracy of 94.88%, an <italic>F</italic><sub>1</sub>-score of 95.18%, and an AUC of 0.9885. The ensemble explanation framework generated robust and clinically meaningful visual explanations in 98.48% of a representative 66-image validation subset used for interpretability assessment.</p><p>These findings demonstrate that combining lightweight deep learning architectures with complementary explainability methods can provide both high diagnostic accuracy and clinical transparency. The proposed framework has strong potential as a decision-support tool for dermatologists, particularly in settings where specialist expertise is limited.</p></sec></sec></body><back><ack><p>We would like to acknowledge King Abdullah University Hospital (KAUH) for their supportive efforts and collaboration. The authors wrote all the contents of the manuscript. Claude Sonnet 4.5 was used to identify any grammar issues or provide sentence update suggestions. The authors updated some sentences according to the suggestions.</p></ack><notes><sec><title>Funding</title><p>This research was funded by Jordan University of Science and Technology (grant number 20230114).</p></sec><sec><title>Data Availability</title><p>The dataset used in this study contains clinical images obtained from King Abdullah University Hospital and includes patient information that is subject to ethical, privacy, and confidentiality restrictions. Therefore, the raw image data are not publicly available. Deidentified data may be made available from the corresponding author upon reasonable request and subject to approval by the relevant institutional review board and data-sharing regulations.</p></sec></notes><fn-group><fn fn-type="con"><p>AAA, DMA, and SAA conceptualized and designed the study. AAA, SAK, and SAA developed the methodology, implemented the proposed framework, conducted the experiments, and performed the data analysis. LZ supervised the research, provided technical guidance on model optimization and validation strategies, and contributed to manuscript structuring. AAA, DMA, SAK, SAA, and BME provided clinical expertise, validated the results, and contributed to the interpretation and medical evaluation of the findings. AAA and SAA prepared the original manuscript draft. LZ and SAK critically reviewed and edited the manuscript. All authors reviewed and approved the final version of the manuscript and agreed to be accountable for all aspects of the work.</p></fn><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb2">CAM</term><def><p>class activation mapping</p></def></def-item><def-item><term id="abb3">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb4">Cycle-GAN</term><def><p>Cycle-Consistent Adversarial Network</p></def></def-item><def-item><term id="abb5">DenseNet</term><def><p>Densely Connected Convolutional Networks</p></def></def-item><def-item><term id="abb6">Grad-CAM</term><def><p>gradient-weighted class activation mapping</p></def></def-item><def-item><term id="abb7">KAUH</term><def><p>King Abdullah University Hospital</p></def></def-item><def-item><term id="abb8">PIH</term><def><p> postinflammatory hypopigmentation</p></def></def-item><def-item><term id="abb9">ResNet</term><def><p>residual network</p></def></def-item><def-item><term id="abb10">ROC</term><def><p>receiver operating characteristic</p></def></def-item><def-item><term id="abb11">SHAP</term><def><p>Shapley additive explanations</p></def></def-item><def-item><term id="abb12">SmoothGrad</term><def><p>smooth gradients</p></def></def-item><def-item><term id="abb13">VGG</term><def><p>Visual Geometry Group</p></def></def-item><def-item><term id="abb14">XAI</term><def><p>explainable artificial intelligence</p></def></def-item><def-item><term id="abb15">YOLO</term><def><p>You Only Look Once</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Patrick</surname><given-names>MT</given-names> </name><name name-style="western"><surname>Sreeskandarajan</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Large-scale epidemiological analysis of common skin diseases to identify shared and unique comorbidities and demographic factors</article-title><source>Front Immunol</source><year>2023</year><volume>14</volume><fpage>1309549</fpage><pub-id pub-id-type="doi">10.3389/fimmu.2023.1309549</pub-id><pub-id pub-id-type="medline">38259463</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ennab</surname><given-names>M</given-names> </name><name name-style="western"><surname>Mcheick</surname><given-names>H</given-names> </name></person-group><article-title>Advancing AI interpretability in medical imaging: a comparative analysis of pixel-level interpretability and Grad-CAM models</article-title><source>Mach Learn Knowl Extr</source><year>2025</year><volume>7</volume><issue>1</issue><fpage>12</fpage><pub-id pub-id-type="doi">10.3390/make7010012</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fatima</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shaikh</surname><given-names>H</given-names> </name><name name-style="western"><surname>Sahito</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kehar</surname><given-names>A</given-names> </name></person-group><article-title>A review of skin disease detection using deep learning</article-title><source>VFAST Trans Softw Eng</source><year>2024</year><volume>12</volume><issue>4</issue><fpage>220</fpage><lpage>238</lpage><pub-id pub-id-type="doi">10.21015/vtse.v12i4.2022</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>C</given-names> </name><name name-style="western"><surname>Gao</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Intelligent diagnosis of hypopigmented dermatoses and intelligent evaluation of vitiligo severity on the basis of deep learning</article-title><source>Dermatol Ther (Heidelb)</source><year>2024</year><month>12</month><volume>14</volume><issue>12</issue><fpage>3307</fpage><lpage>3320</lpage><pub-id pub-id-type="doi">10.1007/s13555-024-01296-9</pub-id><pub-id pub-id-type="medline">39514178</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Xue</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>AI fusion of multisource data identifies key features of vitiligo</article-title><source>Sci Rep</source><year>2024</year><month>10</month><day>16</day><volume>14</volume><issue>1</issue><fpage>24278</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-75062-4</pub-id><pub-id pub-id-type="medline">39414917</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Griffiths</surname><given-names>CEM</given-names> </name><name name-style="western"><surname>Barker</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bleiker</surname><given-names>TO</given-names> </name><name name-style="western"><surname>Hussain</surname><given-names>W</given-names> </name><name name-style="western"><surname>Simpson</surname><given-names>RC</given-names> </name></person-group><source>Rook&#x2019;s Textbook of Dermatology</source><year>2024</year><edition>10</edition><publisher-name>Wiley-Blackwell</publisher-name><pub-id pub-id-type="doi">10.1002/9781119709268</pub-id><pub-id pub-id-type="other">9781119709213</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Kang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Amagai</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bruckner</surname><given-names>AL</given-names> </name></person-group><source>Fitzpatrick&#x2019;s Dermatology</source><year>2019</year><edition>9</edition><publisher-name>McGraw-Hill Education</publisher-name><pub-id pub-id-type="other">9780071837798</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="book"><person-group person-group-type="editor"><name name-style="western"><surname>Nijhawan</surname><given-names>R</given-names> </name><name name-style="western"><surname>Mendirtta</surname><given-names>N</given-names> </name><name name-style="western"><surname>Verma</surname><given-names>M</given-names> </name><name name-style="western"><surname>Bohra</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kumar</surname><given-names>S</given-names> </name></person-group><article-title>Vitiligo detection using machine learning algorithms</article-title><source>Intelligent Sustainable Systems</source><year>2024</year><publisher-name>Springer Nature Singapore</publisher-name><fpage>33</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1007/978-981-99-8031-4_4</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Esteva</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kuprel</surname><given-names>B</given-names> </name><name name-style="western"><surname>Novoa</surname><given-names>RA</given-names> </name><etal/></person-group><article-title>Dermatologist-level classification of skin cancer with deep neural networks</article-title><source>Nature</source><year>2017</year><month>02</month><day>2</day><volume>542</volume><issue>7639</issue><fpage>115</fpage><lpage>118</lpage><pub-id pub-id-type="doi">10.1038/nature21056</pub-id><pub-id pub-id-type="medline">28117445</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>D</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jin</surname><given-names>Y</given-names> </name></person-group><article-title>Interpretability-based multimodal convolutional neural networks for skin lesion diagnosis</article-title><source>IEEE Trans Cybern</source><year>2022</year><month>12</month><volume>52</volume><issue>12</issue><fpage>12623</fpage><lpage>12637</lpage><pub-id pub-id-type="doi">10.1109/TCYB.2021.3069920</pub-id><pub-id pub-id-type="medline">34546933</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tschandl</surname><given-names>P</given-names> </name><name name-style="western"><surname>Rinner</surname><given-names>C</given-names> </name><name name-style="western"><surname>Apalla</surname><given-names>Z</given-names> </name><etal/></person-group><article-title>Human-computer collaboration for skin cancer recognition</article-title><source>Nat Med</source><year>2020</year><month>08</month><volume>26</volume><issue>8</issue><fpage>1229</fpage><lpage>1234</lpage><pub-id pub-id-type="doi">10.1038/s41591-020-0942-0</pub-id><pub-id pub-id-type="medline">32572267</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jain</surname><given-names>A</given-names> </name><name name-style="western"><surname>Eng</surname><given-names>C</given-names> </name><etal/></person-group><article-title>A deep learning system for differential diagnosis of skin diseases</article-title><source>Nat Med</source><year>2020</year><month>06</month><volume>26</volume><issue>6</issue><fpage>900</fpage><lpage>908</lpage><pub-id pub-id-type="doi">10.1038/s41591-020-0842-3</pub-id><pub-id pub-id-type="medline">32424212</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Usman</surname><given-names>M</given-names> </name><name name-style="western"><surname>Iqbal</surname><given-names>MY</given-names> </name><name name-style="western"><surname>Zafar</surname><given-names>K</given-names> </name><name name-style="western"><surname>Basharat</surname><given-names>S</given-names> </name></person-group><article-title>A novel approach to vitiligo diagnosis using artificial neural networks and dermatological image analysis</article-title><source>J Comput Biomed Inform</source><year>2024</year><access-date>2026-07-23</access-date><volume>8</volume><issue>1</issue><comment><ext-link ext-link-type="uri" xlink:href="https://jcbi.org/index.php/Main/article/view/736">https://jcbi.org/index.php/Main/article/view/736</ext-link></comment></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chaddad</surname><given-names>A</given-names> </name><name name-style="western"><surname>Hu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wen</surname><given-names>B</given-names> </name><name name-style="western"><surname>Kateb</surname><given-names>R</given-names> </name></person-group><article-title>Generalizable and explainable deep learning for medical image computing: an overview</article-title><source>Curr Opin Biomed Eng</source><year>2025</year><month>03</month><volume>33</volume><fpage>100567</fpage><pub-id pub-id-type="doi">10.1016/j.cobme.2024.100567</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Dhore</surname><given-names>V</given-names> </name><name name-style="western"><surname>Bhat</surname><given-names>A</given-names> </name><name name-style="western"><surname>Nerlekar</surname><given-names>V</given-names> </name><name name-style="western"><surname>Chavhan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Umare</surname><given-names>A</given-names> </name></person-group><article-title>Enhancing explainable AI: a hybrid approach combining GradCAM and LRP for CNN interpretability</article-title><source>arXiv</source><comment>Preprint posted online on  May 20, 2024</comment><pub-id pub-id-type="doi">10.48550/arXiv.2405.12175</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Selvaraju</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Cogswell</surname><given-names>M</given-names> </name><name name-style="western"><surname>Das</surname><given-names>A</given-names> </name><name name-style="western"><surname>Vedantam</surname><given-names>R</given-names> </name><name name-style="western"><surname>Parikh</surname><given-names>D</given-names> </name><name name-style="western"><surname>Batra</surname><given-names>D</given-names> </name></person-group><article-title>Grad-CAM: visual explanations from deep networks via gradient-based localization</article-title><source>Int J Comput Vis</source><year>2020</year><month>02</month><volume>128</volume><issue>2</issue><fpage>336</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1007/s11263-019-01228-7</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sundararajan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Taly</surname><given-names>A</given-names> </name><name name-style="western"><surname>Yan</surname><given-names>Q</given-names> </name></person-group><article-title>Axiomatic attribution for deep networks</article-title><access-date>2026-07-07</access-date><conf-name>Proceedings of the 34th International Conference on Machine Learning (ICML 2017)</conf-name><conf-date>Aug 6-11, 2017</conf-date><conf-loc>Sydney, New South Wales, Australia</conf-loc><fpage>3319</fpage><lpage>3328</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v70/sundararajan17a/sundararajan17a.pdf">https://proceedings.mlr.press/v70/sundararajan17a/sundararajan17a.pdf</ext-link></comment></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Smilkov</surname><given-names>D</given-names> </name><name name-style="western"><surname>Thorat</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>B</given-names> </name><name name-style="western"><surname>Vi&#x00E9;gas</surname><given-names>F</given-names> </name><name name-style="western"><surname>Wattenberg</surname><given-names>M</given-names> </name></person-group><article-title>SmoothGrad: removing noise by adding noise</article-title><source>arXiv</source><comment>Preprint posted online on  Jun 12, 2017</comment><pub-id pub-id-type="doi">10.48550/arXiv.1706.03825</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Sandler</surname><given-names>M</given-names> </name><name name-style="western"><surname>Howard</surname><given-names>A</given-names> </name><name name-style="western"><surname>Zhu</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zhmoginov</surname><given-names>A</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>LC</given-names> </name></person-group><article-title>MobileNetV2: inverted residuals and linear bottlenecks</article-title><conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name><conf-date>Jun 18-23, 2018</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2018.00474</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Gholijani</surname><given-names>N</given-names> </name><name name-style="western"><surname>Taheri</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Bazgiri</surname><given-names>M</given-names> </name><name name-style="western"><surname>Darya</surname><given-names>G</given-names> </name><name name-style="western"><surname>Dehghan</surname><given-names>Z</given-names> </name></person-group><article-title>Transforming vitiligo diagnosis and treatment through artificial intelligence: a review</article-title><source>Scand J Immunol</source><year>2025</year><month>12</month><volume>102</volume><issue>6</issue><fpage>e70076</fpage><pub-id pub-id-type="doi">10.1111/sji.70076</pub-id><pub-id pub-id-type="medline">41354974</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Guo</surname><given-names>L</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Ding</surname><given-names>H</given-names> </name><etal/></person-group><article-title>A deep learning-based hybrid artificial intelligence model for the detection and severity assessment of vitiligo lesions</article-title><source>Ann Transl Med</source><year>2022</year><month>05</month><volume>10</volume><issue>10</issue><fpage>590</fpage><pub-id pub-id-type="doi">10.21037/atm-22-1738</pub-id><pub-id pub-id-type="medline">35722422</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhong</surname><given-names>F</given-names> </name><name name-style="western"><surname>He</surname><given-names>K</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Optimizing vitiligo diagnosis with ResNet and Swin transformer deep learning models: a study on performance and interpretability</article-title><source>Sci Rep</source><year>2024</year><month>04</month><day>21</day><volume>14</volume><issue>1</issue><fpage>9127</fpage><pub-id pub-id-type="doi">10.1038/s41598-024-59436-2</pub-id><pub-id pub-id-type="medline">38644396</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kantoria</surname><given-names>V</given-names> </name><name name-style="western"><surname>Sharma</surname><given-names>S</given-names> </name><name name-style="western"><surname>Bhushan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Saini</surname><given-names>H</given-names> </name><name name-style="western"><surname>Nijhawan</surname><given-names>R</given-names> </name></person-group><article-title>Implication of convolutional neural network in the classification of vitiligo</article-title><source>Int Res J Eng Tech</source><year>2020</year><access-date>2026-07-07</access-date><volume>7</volume><issue>3</issue><fpage>5170</fpage><lpage>5176</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://www.irjet.net/archives/V7/i3/IRJET-V7I31036.pdf">https://www.irjet.net/archives/V7/i3/IRJET-V7I31036.pdf</ext-link></comment></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Szegedy</surname><given-names>C</given-names> </name><name name-style="western"><surname>Vanhoucke</surname><given-names>V</given-names> </name><name name-style="western"><surname>Ioffe</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shlens</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wojna</surname><given-names>Z</given-names> </name></person-group><article-title>Rethinking the inception architecture for computer vision</article-title><conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 27-30, 2016</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2016.308</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Simonyan</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zisserman</surname><given-names>A</given-names> </name></person-group><article-title>Very deep convolutional networks for large-scale image recognition</article-title><access-date>2026-07-07</access-date><conf-name>3rd International Conference on Learning Representations, ICLR 2015</conf-name><conf-date>May 7-9, 2015</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://bibbase.org/network/publication/simonyan-zisserman-verydeepconvolutionalnetworksforlargescaleimagerecognition-2015">https://bibbase.org/network/publication/simonyan-zisserman-verydeepconvolutionalnetworksforlargescaleimagerecognition-2015</ext-link></comment></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Iandola</surname><given-names>FN</given-names> </name><name name-style="western"><surname>Han</surname><given-names>S</given-names> </name><name name-style="western"><surname>Moskewicz</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Ashraf</surname><given-names>K</given-names> </name><name name-style="western"><surname>Dally</surname><given-names>WJ</given-names> </name><name name-style="western"><surname>Keutzer</surname><given-names>K</given-names> </name></person-group><article-title>SqueezeNet: alexnet-level accuracy with 50x fewer parameters and &#x003C;0.5MB model size</article-title><source>arXiv</source><comment>Preprint posted online on  Feb 24, 2016</comment><pub-id pub-id-type="doi">10.48550/arXiv.1602.07360</pub-id><pub-id pub-id-type="medline">26778305</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>W</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name><name name-style="western"><surname>Huang</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Zhao</surname><given-names>N</given-names> </name></person-group><article-title>An effective vitiligo intelligent classification system</article-title><source>J Ambient Intell Human Comput</source><year>2023</year><month>05</month><volume>14</volume><issue>5</issue><fpage>5479</fpage><lpage>5488</lpage><pub-id pub-id-type="doi">10.1007/s12652-020-02357-5</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>L</given-names> </name><name name-style="western"><surname>Mishra</surname><given-names>S</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>T</given-names> </name><etal/></person-group><article-title>Design and assessment of convolutional neural network based methods for vitiligo diagnosis</article-title><source>Front Med (Lausanne)</source><year>2021</year><volume>8</volume><fpage>754202</fpage><pub-id pub-id-type="doi">10.3389/fmed.2021.754202</pub-id><pub-id pub-id-type="medline">34733869</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Huang</surname><given-names>G</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Van Der Maaten</surname><given-names>L</given-names> </name><name name-style="western"><surname>Weinberger</surname><given-names>KQ</given-names> </name></person-group><article-title>Densely connected convolutional networks</article-title><conf-name>2017 IEEE Conference on Computer Vision and Pattern Recognition</conf-name><pub-id pub-id-type="doi">10.1109/CVPR.2017.243</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Chattopadhay</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sarkar</surname><given-names>A</given-names> </name><name name-style="western"><surname>Howlader</surname><given-names>P</given-names> </name><name name-style="western"><surname>Balasubramanian</surname><given-names>VN</given-names> </name></person-group><article-title>Grad-cam++: generalized gradient-based visual explanations for deep convolutional networks</article-title><conf-name>2018 IEEE Winter Conference on Applications of Computer Vision (WACV)</conf-name><conf-date>Mar 12-15, 2018</conf-date><pub-id pub-id-type="doi">10.1109/WACV.2018.00097</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>H</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Du</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Score-CAM: score-weighted visual explanations for convolutional neural networks</article-title><conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</conf-name><conf-date>Jun 14-19, 2020</conf-date><pub-id pub-id-type="doi">10.1109/CVPRW50498.2020.00020</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Badhon</surname><given-names>SMSI</given-names> </name><name name-style="western"><surname>Khushbu</surname><given-names>SA</given-names> </name><name name-style="western"><surname>Shaqib</surname><given-names>SM</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>MA</given-names> </name><name name-style="western"><surname>Anik</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Hossain</surname><given-names>KSMT</given-names> </name></person-group><article-title>Explainable AI for skin disease classification using gradient-weighted class activation mapping and transfer learning in digital health to identify contours</article-title><source>Digit Health</source><year>2025</year><volume>11</volume><fpage>20552076251404523</fpage><pub-id pub-id-type="doi">10.1177/20552076251404523</pub-id><pub-id pub-id-type="medline">41393849</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Liu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Cao</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Swin transformer: hierarchical vision transformer using shifted windows</article-title><conf-name>2021 IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name><conf-date>Oct 10-17, 2021</conf-date><pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00986</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shorfuzzaman</surname><given-names>M</given-names> </name></person-group><article-title>An explainable stacked ensemble of deep learning models for improved melanoma skin cancer detection</article-title><source>Multimed Syst</source><year>2022</year><month>08</month><volume>28</volume><issue>4</issue><fpage>1309</fpage><lpage>1323</lpage><pub-id pub-id-type="doi">10.1007/s00530-021-00787-5</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Deng</surname><given-names>J</given-names> </name><name name-style="western"><surname>Dong</surname><given-names>W</given-names> </name><name name-style="western"><surname>Socher</surname><given-names>R</given-names> </name><name name-style="western"><surname>Li</surname><given-names>LJ</given-names> </name><name name-style="western"><surname>Li</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fei-Fei</surname><given-names>L</given-names> </name></person-group><article-title>ImageNet: a large-scale hierarchical image database</article-title><conf-name>2009 IEEE Conference on Computer Vision and Pattern Recognition</conf-name><conf-date>Jun 20-25, 2009</conf-date><pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>HE</given-names> </name><name name-style="western"><surname>Cosa-Linan</surname><given-names>A</given-names> </name><name name-style="western"><surname>Santhanam</surname><given-names>N</given-names> </name><name name-style="western"><surname>Jannesari</surname><given-names>M</given-names> </name><name name-style="western"><surname>Maros</surname><given-names>ME</given-names> </name><name name-style="western"><surname>Ganslandt</surname><given-names>T</given-names> </name></person-group><article-title>Transfer learning for medical image classification: a literature review</article-title><source>BMC Med Imaging</source><year>2022</year><month>04</month><day>13</day><volume>22</volume><issue>1</issue><fpage>69</fpage><pub-id pub-id-type="doi">10.1186/s12880-022-00793-7</pub-id><pub-id pub-id-type="medline">35418051</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Molnar</surname><given-names>C</given-names> </name></person-group><source>Interpretable Machine Learning: A Guide for Making Black Box Models Explainable</source><year>2022</year><access-date>2026-07-07</access-date><edition>3</edition><publisher-name>Christoph Molnar</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://christophm.github.io/interpretable-ml-book/">https://christophm.github.io/interpretable-ml-book/</ext-link></comment></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Tan</surname><given-names>M</given-names> </name><name name-style="western"><surname>Le</surname><given-names>QV</given-names> </name></person-group><article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title><access-date>2026-07-07</access-date><conf-name>the 36th International Conference on Machine Learning</conf-name><conf-date>Jun 9-15, 2019</conf-date><comment><ext-link ext-link-type="uri" xlink:href="https://dblp.org/rec/conf/icml/TanL19.html">https://dblp.org/rec/conf/icml/TanL19.html</ext-link></comment></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kazmierczak</surname><given-names>R</given-names> </name><name name-style="western"><surname>Berthier</surname><given-names>E</given-names> </name><name name-style="western"><surname>Frehse</surname><given-names>G</given-names> </name><name name-style="western"><surname>Franchi</surname><given-names>G</given-names> </name></person-group><article-title>Explainability and vision foundation models: a survey</article-title><source>Inf Fusion</source><year>2025</year><month>10</month><volume>122</volume><fpage>103184</fpage><pub-id pub-id-type="doi">10.1016/j.inffus.2025.103184</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>S</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>General lightweight framework for vision foundation model supporting multi-task and multi-center medical image analysis</article-title><source>Nat Commun</source><year>2025</year><volume>16</volume><issue>1</issue><fpage>2097</fpage><pub-id pub-id-type="doi">10.1038/s41467-025-57427-z</pub-id><pub-id pub-id-type="medline">40025028</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Training and validation curves for all 5 cross-validation folds during the frozen training and fine-tuning phases, including fold-specific best-validation and stopping epochs.</p><media xlink:href="medinform_v14i1e81942_app1.docx" xlink:title="DOCX File, 528 KB"/></supplementary-material></app-group></back></article>