<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">JMIR Med Inform</journal-id><journal-id journal-id-type="publisher-id">medinform</journal-id><journal-id journal-id-type="index">7</journal-id><journal-title>JMIR Medical Informatics</journal-title><abbrev-journal-title>JMIR Med Inform</abbrev-journal-title><issn pub-type="epub">2291-9694</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v14i1e91960</article-id><article-id pub-id-type="doi">10.2196/91960</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>R-R Interval Histogram-Based Deep Learning for 3-Class Atrial Fibrillation Screening in Garment-Type Wearable Holter Electrocardiogram Monitoring: Algorithm Development and Validation Study</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Nakano</surname><given-names>Tomoaki</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Okayama</surname><given-names>Keita</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yasuoka</surname><given-names>Masanao</given-names></name><degrees>BE</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Shigeta</surname><given-names>Hironori</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Kawamura</surname><given-names>Akito</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sekihara</surname><given-names>Takayuki</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Ozu</surname><given-names>Kentaro</given-names></name><degrees>MD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Oka</surname><given-names>Takafumi</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Seno</surname><given-names>Shigeto</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Sakata</surname><given-names>Yasushi</given-names></name><degrees>MD, PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib></contrib-group><aff id="aff1"><institution>Department of Cardiovascular Medicine, Graduate School of Medicine, The University of Osaka</institution><addr-line>2-2, Yamadaoka</addr-line><addr-line>Suita</addr-line><addr-line>Osaka</addr-line><country>Japan</country></aff><aff id="aff2"><institution>Department of Bioinformatic Engineering, Graduate School of Information Science and Technology, The University of Osaka</institution><addr-line>Suita</addr-line><addr-line>Osaka</addr-line><country>Japan</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Coristine</surname><given-names>Andrew</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Tanaka</surname><given-names>Motoshi</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Krasteva</surname><given-names>Vessela</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Keita Okayama, MD, PhD, Department of Cardiovascular Medicine, Graduate School of Medicine, The University of Osaka, 2-2, Yamadaoka, Suita, Osaka, 565-0871, Japan, 81 06-6879-3632; <email>okayama@cardiology.med.osaka-u.ac.jp</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>24</day><month>7</month><year>2026</year></pub-date><volume>14</volume><elocation-id>e91960</elocation-id><history><date date-type="received"><day>22</day><month>01</month><year>2026</year></date><date date-type="rev-recd"><day>14</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>16</day><month>06</month><year>2026</year></date></history><copyright-statement>&#x00A9; Tomoaki Nakano, Keita Okayama, Masanao Yasuoka, Hironori Shigeta, Akito Kawamura, Takayuki Sekihara, Kentaro Ozu, Takafumi Oka, Shigeto Seno, Yasushi Sakata. Originally published in JMIR Medical Informatics (<ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org">https://medinform.jmir.org</ext-link>), 24.7.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in JMIR Medical Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://medinform.jmir.org/">https://medinform.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://medinform.jmir.org/2026/1/e91960"/><abstract><sec><title>Background</title><p>Long-term garment-type wearable Holter electrocardiographic (ECG) monitoring is frequently affected by noise contamination, which complicates automated atrial fibrillation (AF) detection in real-world recordings. Although deep learning has shown high performance for AF detection, relatively few studies have evaluated explicit strategies for handling noise-included wearable ECG data. An alternative representation using the R-R interval (RRI) time series may reduce the dependence on waveform morphology and provide an alternative pathway for AF screening in noisy recordings.</p></sec><sec><title>Objective</title><p>This study aimed to develop and evaluate a 3-class, noise-aware RRI-based AF screening framework that explicitly separated AF, non-AF, and uninterpretable noise windows, and to assess the impact of analysis window length on model performance.</p></sec><sec sec-type="methods"><title>Methods</title><p>Single-lead garment-type wearable Holter ECG data from 117 patients at the University of Osaka Hospital were analyzed after exclusion of patients with documented atrial tachycardia, flutter, or paced rhythm according to the predefined task definition. R-peaks were automatically detected, and the resulting RRI segments were converted into 2D histogram images, with time on the x-axis and RRI-derived heart rate on the y-axis, for 1.5-, 3-, and 6-minute windows. A ResNet-34&#x2013;based 2D convolutional neural network was trained for 3-class classification. Model performance was evaluated using 5-fold interpatient cross-validation on the institutional dataset and independent external testing on the MIT-BIH (Massachusetts Institute of Technology&#x2013;Beth Israel Hospital) AF Database (AFDB). In the external validation, atrial flutter&#x2013;annotated intervals were excluded to match the training task definition. Patient-level AF burden was evaluated by comparing reference AF burden with model-estimated AF burden using Pearson and Spearman correlation coefficients, and linear regression.</p></sec><sec sec-type="results"><title>Results</title><p>Of 129 monitored patients between March 1, 2023, and November 20, 2025, 117 were analyzed. In the internal validation, the 3-class model (non-AF, AF, and noise) showed similarly high performance for the 1.5- and 3-minute windows, both with an accuracy of 96.6%. In independent external validation, the 3-minute window showed numerically the highest overall performance (accuracy: 97.3%; AF sensitivity: 96.9%; and AF specificity: 97.7%), although the differences across window lengths were modest. At the patient level, AF burden correlation was high across all window lengths, with Pearson <italic>r</italic> of 0.995, 0.991, and 0.989 and Spearman &#x03C1; of 0.988, 0.982, and 0.979 for the 1.5-, 3-, and 6-minute models, respectively.</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>The RRI-based 2D convolutional neural network achieved high AF classification accuracy and strong patient-level correlation with reference AF burden. Using RRI features and a 3-class framework, which explicitly separated noise from AF and non-AF rhythms, a 3-minute RRI window provided a favorable balance of performance for AF screening in a garment-type Holter ECG.</p></sec></abstract><kwd-group><kwd>atrial fibrillation</kwd><kwd>deep neural network</kwd><kwd>paroxysmal atrial fibrillation</kwd><kwd>long-term Holter electrocardiogram</kwd><kwd>R-R interval</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Atrial fibrillation (AF) is a common arrhythmia and a major cause of cardioembolic stroke [<xref ref-type="bibr" rid="ref1">1</xref>,<xref ref-type="bibr" rid="ref2">2</xref>]. Early and accurate detection of paroxysmal AF (PAF), which may manifest asymptomatically or transiently, is important for timely anticoagulation and rhythm control strategies. Given the unpredictable onset of PAF, long-term Holter electrocardiographic (ECG) monitoring has been extensively used to capture AF episodes [<xref ref-type="bibr" rid="ref3">3</xref>-<xref ref-type="bibr" rid="ref7">7</xref>]. Garment-type wearable Holter ECG devices, which integrate electrodes into shirts or belts, have garnered attention for their ability to facilitate long-term continuous recording during daily activities with minimal burden on the wearer [<xref ref-type="bibr" rid="ref8">8</xref>-<xref ref-type="bibr" rid="ref10">10</xref>].</p><p>In garment-type wearable Holter ECG monitoring, the characteristics of noise contamination differ from those of conventional adhesive or patch-type recordings. As the garment can be removed and reworn, recordings may contain not only intermittent contact artifacts during wear, caused by body motion, variable electrode contact, and changes in skin-electrode impedance, but also off-body periods, signal loss intervals, and segments acquired during unstable reattachment [<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref11">11</xref>-<xref ref-type="bibr" rid="ref13">13</xref>]. These heterogeneous segments are ultimately included in the long-term recordings submitted for clinical review, making manual analysis by clinical technologists or physicians time-consuming and labor-intensive [<xref ref-type="bibr" rid="ref3">3</xref>,<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. Therefore, the practical challenge in diagnostic support using automated analysis is not limited to improving AF detection accuracy in relatively low-noise ECG segments but also includes avoiding forced classification into AF or non-AF in windows in which rhythm interpretation is unreliable. Recent studies, including the PhysioNet or Computing in Cardiology Challenge 2017, have emphasized the importance of AF classification from noisy single-lead ECG recordings and have advanced algorithm development in this setting [<xref ref-type="bibr" rid="ref7">7</xref>,<xref ref-type="bibr" rid="ref16">16</xref>,<xref ref-type="bibr" rid="ref17">17</xref>]. However, corresponding evidence in long-term garment-type wearable Holter ECG remains limited.</p><p>Automatic ECG analysis using AI, particularly deep learning (DL), has recently emerged as a promising approach to address these challenges. Although DL architectures that directly process raw ECG waveforms, including 1D convolutional neural networks (1D-CNNs) [<xref ref-type="bibr" rid="ref14">14</xref>-<xref ref-type="bibr" rid="ref18">18</xref>], have demonstrated high performance, their performance may deteriorate in the presence of waveform noise in wearable recordings. In contrast, approaches based on the R-R interval (RRI) focus on the beat-to-beat irregularity, which is fundamental to AF and may be relatively tolerant of waveform contamination [<xref ref-type="bibr" rid="ref19">19</xref>-<xref ref-type="bibr" rid="ref22">22</xref>]. Moreover, converting the RRI time series into 2D images allows AF rhythm characteristics to be represented as structured visual patterns, making this approach highly compatible with analyses based on 2D convolutional neural networks (2D-CNNs) [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. In noise-controlled patch-type Holter recordings, such RRI-based 2D-CNN approaches have achieved high accuracy with short segment windows of approximately 90 beats [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref24">24</xref>].</p><p>However, a suitable RRI-based DL framework for noise-prone garment-type monitoring remains unclear. In particular, it is uncertain how analysis window length influences diagnostic performance when analyzable rhythm portions and uninterpretable artifact-contaminated portions coexist within long-term recordings. Therefore, this study aimed to develop and evaluate an RRI histogram-based 3-class AF screening framework for wearable Holter ECG recordings, in which visually uninterpretable or unreliable segments were treated as an independent noise class, separate from AF and non-AF rhythms. By examining multiple analysis window lengths, we sought to identify an approach that could support efficient and reliable AF screening in real-world wearable monitoring.</p></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><sec id="s2-1-1"><title>Task Definition</title><p>This study aimed to develop a machine learning framework for AF screening based on RRI irregularity while explicitly separating uninterpretable segments as noise. Atrial tachycardia (AT) and atrial flutter (AFL) typically exhibit regular RRIs and lack the irregular characteristics of AF, thus representing distinct physiological rhythmic phenotypes [<xref ref-type="bibr" rid="ref26">26</xref>]. Therefore, AT/AFL episodes were excluded from both the training and validation datasets to preserve the clinical coherence of the task rather than to optimize model performance. The diagnosis and characterization of AT/AFL constitute a separate clinical challenge and are considered beyond the scope of this analysis.</p><p>In addition, this study was designed to systematically evaluate the impact of window length as part of a sensitivity analysis of the performance of the RRI-based AF detection model in the context of long-term garment-type wearable Holter ECG monitoring, which is susceptible to noise contamination. Therefore, noise was treated as a distinct label to reflect real-world data quality and enable the model to explicitly separate uninterpretable segments from AF and non-AF rhythms. Three window lengths were examined: 1.5-, 3-, and 6-minute windows.</p></sec><sec id="s2-1-2"><title>Study Population</title><p>The study population comprised patients who underwent monitoring using a wearable Holter ECG device (PS201-01; Mitsufuji) at the University of Osaka Hospital between March 1, 2023, and November 20, 2025. Examinations were performed to evaluate suspected or known arrhythmias or symptoms such as palpitations.</p></sec><sec id="s2-1-3"><title>Ethical Considerations</title><p>This study was approved by the Ethics Committee of the University of Osaka Hospital on March 1, 2023 (approval 22449) and was conducted in accordance with the Declaration of Helsinki. Written informed consent was obtained from all participants prior to study enrollment. All study data were deidentified before analysis to protect participant privacy and confidentiality. No financial compensation or other incentives were provided to the participants. No images in the manuscript or supplementary material contain information that could identify individual participants.</p></sec></sec><sec id="s2-2"><title>ECG Recording and Data Collection</title><p>ECG data were acquired using a PS201-01 wearable Holter ECG device [<xref ref-type="bibr" rid="ref10">10</xref>] with either shirt-type or belt-type garment electrodes (<xref ref-type="fig" rid="figure1">Figure 1</xref>).</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>Configuration of the garment-type wearable Holter ECG system. The shirt-type garment (left) uses electrodes incorporating silver fiber AGposs, whereas the belt-type garment (right) uses electrodes incorporating both AGposs and the silver-containing paste DuraQ. The orange frames indicate the electrode positions. The transmitter was attached within the white dotted-line area and transmitted electrocardiogram data via Bluetooth to a smartphone app, where the signals were stored.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig01.png"/></fig><sec id="s2-2-1"><title>Device Configuration</title><p>The PS201-01 system uses silver fibers (AGposs; Mitsufuji) as the electrode material. The shirt-type garment used AGposs alone, whereas the belt-type garment combined AGposs with electrodes made from a silver-containing paste (DuraQ; Sumitomo Bakelite). Both garments share an equivalent electrode configuration, enabling the acquisition of a single-lead ECG corresponding to lead I of a standard 12-lead ECG.</p></sec><sec id="s2-2-2"><title>Data Acquisition Specifications</title><p>The ECG waveforms acquired using the PS201-01 device were transmitted via Bluetooth to a dedicated smartphone app (wearable Holter ECG management; Mitsufuji) and stored in the Medical Waveform Format Encoding Rules format. The device is a class II long-term ECG recording system approved under the Japanese Pharmaceuticals and Medical Devices Act. The ECG signal was directly sampled by the built-in analog-to-digital converter at 250 Hz and stored at the same sampling frequency without downsampling; no analog-to-digital converter (ADC) internal digital filter or digital low-pass filter was applied before storage. According to manufacturer-provided circuit simulation data, the analog front end was designed to provide band limiting for antialiasing before ADC sampling. The relative gain compared with the passband was &#x2212;2 dB at 100 Hz, &#x2212;4.5 dB at 125 Hz, &#x2212;10 dB at 250 Hz, &#x2212;20 dB at 700 Hz, and &#x2212;33 dB at 1000 Hz; the &#x2212;3 dB, &#x2212;20 dB, and &#x2212;40 dB frequencies were approximately 105 Hz, 700 Hz, and 1.25 kHz, respectively. The sampling frequency of 250 Hz was identical to that used in the MIT-BIH AFDB.</p></sec></sec><sec id="s2-3"><title>Data Preprocessing and Annotation</title><p>For model development, the RRI segments were extracted from raw ECG waveforms and annotated by a cardiologist.</p><sec id="s2-3-1"><title>RRI Data Generation</title><p>For ECG data stored in the Medical Waveform Format Encoding format, an automatic R-peak detection algorithm based on the Hamilton method [<xref ref-type="bibr" rid="ref27">27</xref>], as implemented in the Python BioSPPy library, was applied to generate the RRI time series data. No subsample interpolation, dedicated artifact-rejection procedure, or additional peak refinement technique was applied after automatic detection. Baseline correction was performed using a bandpass filter as a part of the preprocessing pipeline. In noise-labeled segments, R-peak detection was not uniformly reliable and could result in excessive false detections, absent or insufficient detections, or temporally heterogeneous detection patterns, depending on the type and severity of artifact. As the accuracy of R-peak detection critically influences model performance in RRI-based AF diagnosis, this step serves as the foundation of the entire analysis workflow. To support the practical validity of this preprocessing step under 250 Hz sampling, we performed a supplementary quality assessment of R-peak detection described in the following section.</p></sec><sec id="s2-3-2"><title>Quality Assessment of R-Peak Detection</title><p>As a supplementary quality check of the preprocessing pipeline, we performed a manual review of randomly sampled 30-second ECG segments from the institutional dataset used for model development and internal validation. This assessment was performed to evaluate the practical reliability of the Hamilton-based R-peak detection step in the same data source used for training and testing, but it was not used to select, exclude, or reassign segments for model training or testing. All windows were handled according to the predefined label aggregation rules and patient-level cross-validation splits, regardless of the quality assessment results. The review set consisted of 100 pure non-AF segments, 100 pure AF segments, 50 pure noise segments, and 50 noise-containing segments. A cardiologist visually reviewed each segment to determine whether reliable reference R-peaks could be identified. A segment was defined as <italic>annotatable</italic> when R-peaks could be reliably confirmed for at least 20 of the 30 seconds. For annotatable segments, automatic R-peak detection was graded as <italic>acceptable</italic> when &#x003E;90% of automatically detected peaks appropriately corresponded to visually identified R waves, having <italic>minor errors</italic> when the proportion was &#x003E;70% to &#x2264;90%, and <italic>unreliable</italic> otherwise. For noise and noise-containing segments, readability was also recorded.</p></sec><sec id="s2-3-3"><title>Annotation Procedure</title><p>The generated RRI plots and the corresponding raw ECG waveforms were visualized using an in-house ECG annotation tool. A cardiologist reviewed the continuous ECG recordings in chronological order and assigned labels at 10-second intervals, taking the surrounding rhythm context into account. AF was defined as a rhythm showing irregular RRIs without discernible P waves, occupying more than half of a given 10-second interval. When multiple rhythms occurred within a single interval, the rhythm occupying the majority of the interval was assigned as the representative label. Thus, transitions between rhythms were handled by majority labeling within each 10-second interval.</p><p>A cardiologist classified the data into 3 classes:</p><list list-type="bullet"><list-item><p>Non-AF: normal sinus rhythm and non-AF arrhythmias, excluding AT/AFL</p></list-item><list-item><p>AF: atrial fibrillation</p></list-item><list-item><p>Noise: a 10-second interval in which reliable rhythm interpretation was not possible because of severe artifact, unstable electrode contact, signal loss, or periods during which the garment was not worn</p></list-item></list><p>The following ECG patterns were excluded from the analysis: regular supraventricular tachycardia (AT/AFL) and segments containing paced waveforms.</p><p>Annotations were created independently of model training and evaluation. To further assess annotation robustness, supplementary intrarater and independent interrater agreement analyses were performed, as described in the &#x201C;Annotation Agreement Assessment&#x201D; section.</p></sec><sec id="s2-3-4"><title>Annotation Agreement Assessment</title><p>We performed a supplementary annotation verification analysis using randomly sampled ECG segments from the institutional dataset used for model development and internal validation. A custom program was developed to randomly sample segments and display the corresponding ECG data for manual review. Using this procedure, 150 segments were randomly selected, comprising 50 non-AF, 50 AF, and 50 noise segments according to the original study annotations. The sampled segments were then reassessed in 2 ways: first, by the cardiologist who had created the original annotations, using the dedicated review program, and second, by an additional independent cardiologist. Agreement of each reassessment with the original study annotations was evaluated using percentage agreement and Cohen &#x03BA;. This supplementary analysis was designed to examine the robustness of the original annotations and was not used to select, exclude, relabel, or reassign segments for model training or testing. The original study annotations remained the reference labels for model development and evaluation, and all windows were handled according to the predefined label aggregation rules and patient-level cross-validation splits.</p></sec><sec id="s2-3-5"><title>Patient-Level Rhythm Burden Assessment</title><p>To characterize the rhythm composition of the institutional cohort, patient-level AF burden was calculated using the original 10-second annotations. AF burden was defined as the proportion of AF-labeled intervals among analyzable intervals after excluding noise-labeled intervals from the denominator. The distribution of AF burden across patients was summarized to assess whether the cohort contained a broad range of PAF patterns or was dominated by sustained AF and non-AF recordings.</p><p>To address the concern that the non-AF class might consist predominantly of clean sinus rhythm, we performed a rudimentary ectopic burden estimate in patients with an AF burden of 0%. This subgroup was selected because all analyzable intervals in these patients were annotated as non-AF, allowing assessment of ectopy in non-AF recordings without contamination by AF-to-non-AF transition periods. As exhaustive ectopic beat annotation across the full dataset was not feasible and could be affected by noise, representative ECG segments were sampled for manual review. For each eligible patient, six 3-minute segments were selected from the available analyzable non-AF recordings, and ectopic beats were visually identified. Ectopic burden was calculated as the proportion of ectopic beats among all visually assessable beats in the reviewed segments. Patients were categorized according to estimated ectopic burden as &#x003C;1%, 1% to &#x003C;5%, 5% to &#x003C;10%, or &#x2265;10%. This analysis was intended to provide a descriptive estimate of ectopy within the non-AF class rather than a comprehensive beat-level annotation of the entire dataset.</p></sec></sec><sec id="s2-4"><title>Feature Extraction and Training Data Generation</title><p>Two-dimensional feature images used as inputs to the DL model were generated from the RRI time series data to construct the training and evaluation datasets.</p><sec id="s2-4-1"><title>Segmentation of RRI Data</title><p>The RRI time series data were segmented into fixed-length analysis windows. Window lengths of 1.5, 3, and 6 minutes were used, and model performance was evaluated for each setting. These window lengths were selected to cover clinically and methodologically relevant time scales. A 1.5-minute window approximately corresponds to 90 beats at a heart rate of 60 beats per minute, which is similar to the segment length used in prior Lorenz/Poincar&#x00E9; plot&#x2013;based AF detection studies [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. A 6-minute window was included because 6 minutes have been used as a clinically relevant threshold in studies of atrial high-rate episodes [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. A 3-minute window was evaluated as an intermediate setting to assess the trade-off between temporal resolution with shorter windows and classification stability with longer windows.</p></sec><sec id="s2-4-2"><title>Data Augmentation (Sliding Window)</title><p>To improve model generalization and increase the volume of training data, a sliding window approach was applied to all window length settings (<xref ref-type="fig" rid="figure2">Figure 2</xref>). To standardize the augmentation strategy across models while preserving a similar relative degree of overlap, the step size was set to one-third of the target window length. Accordingly, step sizes of 30 seconds, 1 minute, and 2 minutes were used for the 1.5-minute, 3-minute, and 6-minute models, respectively. Cross-validation was performed per patient to prevent interpatient data leakage caused by the sliding window procedure. Thus, overlapping windows derived from the same patient were contained within the same data split and were not shared across training, validation, and test sets. No additional procedure was applied to reduce correlation among overlapping samples within the same split; rather, the sliding window approach was used as an intended augmentation strategy within the training framework.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Schematic overview of sliding window data augmentation and segment-level label assignment used for model development. Ground-truth annotations were provided at 10-s intervals and aggregated to a single segment label (36 labels for a 6-min window; 18 labels for a 3-min window; and 9 labels for a 1.5-min window). Segments containing both atrial fibrillation (AF) and non-AF labels were excluded from the training or validation dataset but handled according to the test time rule (The detailed label aggregation rules are provided in the &#x201C;Label Aggregation Rules&#x201D; subsection).</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig02.png"/></fig></sec><sec id="s2-4-3"><title>Conversion to 2D Images (Histogram Processing)</title><p>Rather than using the raw 1D RRI sequence directly as input, each segment was transformed into a 2D histogram image to represent rhythm irregularity in a structured form. AF is characterized not only by variability in interval length itself but also by the temporal pattern and density distribution of beat-to-beat fluctuations. A 2D representation was therefore adopted to capture both the overall distribution of interval-derived values and their sequential variation, conceptually similar to Lorenz/Poincar&#x00E9;-based rhythm analysis [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>]. In addition, this representation allowed noisy or visually uninterpretable segments to appear as scattered and unstable patterns. Histogram binning aggregates neighboring values and may reduce the influence of minor timing fluctuations or isolated R-peak detection errors. Heat maps were generated as 2D histograms from each RRI segment using the window-level R-peak positions on the x-axis and heart rate derived from the RRI on the y-axis. Specifically, heart rate was calculated as 60,000 divided by the RRI in milliseconds, and the histogram was computed as a count-based density map without additional count normalization. Thus, the present input representation was a time&#x2013;heart rate density histogram rather than a recurrence plot or a &#x0394;RRI map. The y-axis represented RRI-derived heart rate and was predefined to range from 30 to 400 beats per minute, uniformly divided into 200 bins for all window lengths. Thus, RRIs longer than 2.0 seconds, corresponding to heart rates below 30 beats per minute, and extremely short RRIs below 150 milliseconds, corresponding to heart rates above 400 beats per minute, were outside the predefined heart rate range of the histogram representation. These out-of-range intervals were not encoded as separate overflow bins or dedicated features in the model input. For the x-axis, the number of bins was scaled according to segment duration to maintain comparable temporal density across models: 25 bins for 1.5-minute windows, 50 bins for 3-minute windows, and 100 bins for 6-minute windows. Accordingly, the resulting histogram image sizes were 25&#x00D7;200, 50&#x00D7;200, and 100&#x00D7;200 bins, respectively. Histogram counts were used as pixel intensities, and no additional smoothing or interpolation was applied before input into the 2D-CNN. An enlarged representative RRI histogram image is provided in Figure S1 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> to improve visual interpretability of the histogram format. As noise-labeled segments were defined by visually uninterpretable or nondiagnostic ECG quality rather than by a single R-peak detection failure mode, their RRI-derived histograms were heterogeneous. Representative noise-labeled examples included excessive false R-peak detections, absent or insufficient detections, and temporally heterogeneous artifact patterns within the same window. Real ECG waveforms with automatically detected R-peak positions and the corresponding RRI-derived heart rate histograms are shown in Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p></sec><sec id="s2-4-4"><title>Label Aggregation Rules</title><p>As annotations were assigned at 10-second intervals, each segment corresponded to 9 labels for a 1.5-minute window, 18 labels for a 3-minute window, and 36 labels for a 6-minute window. The final window label was determined using separate rule sets for the training/validation and test datasets, as illustrated in <xref ref-type="fig" rid="figure3">Figure 3</xref>.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Window-level ground-truth label assignment rules for the 3-class classification task. Annotations were assigned in 10-s intervals. Consequently, each segment corresponded to 9 labels for a 1.5-min window, 18 labels for a 3-min window, and 36 labels for a 6-min window. The figure illustrates the rules for handling windows containing multiple labels, together with representative electrocardiogram (ECG) examples for both training/validation and test data. Ten-second labels are indicated by background color: red for atrial fibrillation (AF), gray for noise, and white for non-AF. The Noise example shown in the figure represents an actual artifact-contaminated waveform recorded using the garment-type wearable Holter ECG device, not simulated Gaussian noise.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig03.png"/></fig></sec></sec><sec id="s2-5"><title>Training or Validation Data</title><p>The data used for training or validation were handled according to the following rules:</p><list list-type="bullet"><list-item><p>For supervised training, windows containing both AF and non-AF labels were excluded because assigning a single representative label to such windows would introduce ambiguous supervisory signals. This rule was adopted to preserve clear label definitions during model training rather than to deny the existence of rhythm transitions in real-world recordings. This approach may reduce the representation of transitional or mixed rhythm segments in the training dataset and is therefore considered a limitation of the present framework.</p></list-item><list-item><p>For supervised training, windows containing both AF and non-AF labels were excluded because assigning a single representative label to such windows would introduce ambiguous supervisory signals. This rule was adopted to preserve clear label definitions during model training rather than to deny the existence of rhythm transitions in real-world recordings. This approach may reduce the representation of transitional or mixed rhythm segments in the training dataset and is therefore considered a limitation of the present framework.</p></list-item></list></sec><sec id="s2-6"><title>Test Data</title><p>The data used for test were handled according to the following rules:</p><list list-type="bullet"><list-item><p>The most frequently used label within a window (majority vote) was used as the reference label.</p></list-item><list-item><p>In the event of a tie, the reference label was determined using a predefined priority order (non-AF &#x003E; AF &#x003E; noise) solely to ensure deterministic label assignment. This priority order was an operational convention for uncommon edge cases and was not intended to represent a clinically motivated hierarchy among rhythm classes.</p></list-item></list></sec><sec id="s2-7"><title>DL Model and Validation Protocol</title><p>A supervised DL model was constructed to classify the generated 2D RRI images into non-AF, AF, and noise classes.</p><sec id="s2-7-1"><title>DL Model</title><p>We adopted ResNet-34, a 2D-CNN architecture with residual learning that enables stable training in deep networks and has shown strong performance in image recognition [<xref ref-type="bibr" rid="ref30">30</xref>]. ResNet-34 was selected as a practical compromise between representational capacity and model complexity within the ResNet family. In the context of the present wearable ECG dataset, we considered shallower models potentially insufficient to capture the diversity of rhythm patterns, whereas deeper models could increase the risk of overfitting without a clear advantage. To accommodate the present input representation, the initial convolutional layer was modified to accept single-channel inputs, and the final pooling layer was replaced with an adaptive global average pooling layer. As a result, the model could accept the native histogram image sizes (25&#x00D7;200, 50&#x00D7;200, and 100&#x00D7;200) without resizing to a fixed square resolution. The model was initialized from scratch, and no pretrained weights were used. No additional systematic hyperparameter optimization was performed beyond the predefined training settings and early stopping. The model was trained to classify the histogram images derived from the RRI segments into 3 classes: non-AF, AF, and noise. Training was performed using the Adam optimizer (learning rate 1&#x00D7;10&#x207B;&#x2074;), with cross-entropy loss as the objective function.</p></sec><sec id="s2-7-2"><title>Internal Validation</title><p>Internal validation was performed using 5-fold cross-validation on an institutional dataset. To avoid leakage of patient information between the training and test sets, we used an interpatient paradigm, splitting data at the patient level. Patients were randomly divided into 5 mutually exclusive groups based on their identifiers. For each fold, approximately one-fifth of the patients were assigned to the test set, and the remaining four-fifths were used for model development, with the development set further divided into training and validation subsets at a 9:1 ratio. The training split within the nontest patients was also performed at the patient level by random sampling without stratification; therefore, all sliding window samples derived from a given patient were assigned exclusively to either the training or validation subset. No explicit balancing was applied for patient characteristics, such as age, AF burden, or signal quality across folds. The training and evaluation were repeated for each fold using the same procedure. Given the relatively limited size of the institutional dataset, early stopping was used as the primary measure to reduce overfitting. Specifically, training was halted if the validation loss did not improve for 5 consecutive epochs. No additional formal hyperparameter search was performed. As an exploratory post hoc subgroup analysis by garment type, we evaluated the 3-minute model using the same 5-fold cross-validation splits as those used in the primary internal validation. As the shirt-type subgroup had a limited sample size and included few AF cases, separate garment-specific model training was not performed. Instead, in each fold, the model was trained using mixed shirt- and belt-type recordings, and pooled out-of-fold predictions from the corresponding test folds were stratified by garment type.</p></sec><sec id="s2-7-3"><title>External Validation</title><p>To evaluate the generalization performance of the predefined model framework, external validation was conducted using the MIT-BIH AFDB, a widely used benchmark for AF detection. The AFDB was completely independent of the institutional dataset and was used exclusively as an external test dataset; it was not used for model training, validation, or tuning. The AFDB contains 25 two-channel ECG recordings, each approximately 10 hours in duration and sampled at 250 Hz. Two recordings (00735 and 03665) were excluded due to missing signals. As the sampling frequency of 250 Hz matched that of the wearable Holter device, resampling was not required. To match the single-lead configuration of the wearable device, only the channel labeled ECG1 was used for all AFDB recordings; the 2 channels were neither averaged nor randomly selected.</p><p>The AFDB contains annotations comprising 4 rhythm categories: atrial fibrillation (AFIB), atrial flutter (AFL), junctional rhythm (JR), and normal rhythm (NR). Each event was annotated with a sample-level onset position and corresponding label. For comparison with the proposed method, these annotations were converted into 10-second interval labels. When multiple labels occurred within a 10-second interval, the label with the longest duration was assigned as the representative label.</p><p>As AT/AFL episodes were excluded from the institutional training and validation datasets, AFL-annotated intervals were excluded from the primary external validation analysis to ensure consistency between the training task definition and the external test framework. Intervals corresponding to AFIB were mapped to the AF class, whereas those corresponding to JR and NR were mapped to the non-AF class. No additional supraventricular rhythm categories were separately annotated in the AFDB; therefore, no further rhythm-specific remapping was performed. As the database lacked explicit noise labels, intervals containing visually apparent noise were newly annotated and assigned to the noise class. Noise was defined as a visually uninterpretable ECG segment in which reliable rhythm assessment was not possible because of artifact. These annotations were confirmed by 2 cardiologists. As a secondary analysis, we also evaluated an alternative framework in which AFL intervals were mapped to the AF class, reflecting a broader clinical grouping of atrial tachyarrhythmias.</p><p>For external validation, the final model was retrained using the full institutional development dataset with the same predefined architecture and training settings used in the internal validation. A 9:1 training:validation split was used for early stopping only. This approach was chosen so that all available institutional development data contributed to the final model before testing on the completely independent AFDB. The same early stopping strategy was used, in which training was halted if the validation loss did not improve for 5 consecutive epochs. This model was then used to segment the RRI data from the AFDB into segments of each window length, convert them into 2D images, and generate predictions. The same core preprocessing pipeline was used for both the institutional data and AFDB, including R-peak detection from the full recording and generation of window-level RRI histogram images. The main AFDB-specific preprocessing steps were exclusion of nontarget intervals before window generation and remapping of the public annotation labels as described earlier. The reference label for each window was defined as the label of the longest rhythmic region in that window.</p><p>To evaluate the clinical relevance of aggregated model outputs across an entire recording, we assessed the correlation between the reference AF burden and the model-estimated AF burden on a per-patient basis. In the external validation implementation, AFL-annotated intervals were first excluded in accordance with the primary task definition. The reference AF burden was then calculated for each patient, based on the AFDB annotations, as the proportion of AF-labeled 10-second intervals among analyzable intervals after excluding noise-labeled intervals from the denominator. Model-estimated AF burden was similarly calculated as the proportion of windows classified as AF among windows classified as either AF or non-AF, thereby excluding windows classified as noise from the denominator. Mixed-rhythm windows were handled according to the predefined label assignment and window aggregation rules and were not subjected to an additional exclusion step during patient-level burden calculation.</p></sec></sec><sec id="s2-8"><title>Accuracy Evaluation and Statistical Analysis</title><sec id="s2-8-1"><title>Diagnostic Performance Metrics</title><p>Model performance was evaluated at the window level using overall accuracy and class-wise sensitivity, specificity, positive predictive value (PPV), negative predictive value (NPV), <italic>F</italic><sub>1</sub>-score, and area under the receiver operating characteristic curve (AUROC) for the 3 classes (AF, non-AF, and noise). Overall accuracy was defined as the proportion of correctly classified windows among all evaluated windows. For each class, sensitivity, specificity, PPV, NPV, and <italic>F</italic><sub>1</sub>-score were calculated using a one-versus-rest framework, in which the target class was treated as positive and the other 2 classes were combined as negative. For the internal validation, these metrics were calculated separately for each of the 5 cross-validation test folds and then summarized across folds. Macroaveraged <italic>F</italic><sub>1</sub>-score and macroaveraged AUROC were additionally calculated to provide an overall assessment that was less affected by class imbalance. AUROC was calculated using a one-versus-rest approach for each class.</p><p>For AF burden, the association between reference and model prediction was assessed at the patient level. Pearson correlation coefficient was used as the primary measure to assess the linear association between the reference and model-estimated AF burden values. As AF burden may show a skewed distribution, Spearman rank correlation coefficient was additionally calculated as a supplementary nonparametric measure. Linear regression was also performed to characterize the relationship between the 2 methods.</p></sec><sec id="s2-8-2"><title>Statistical Analysis Environment</title><p>All analyses were performed using Python 3.8.8 on the Windows 11 platform. The computational environment comprised a 3.6 GHz 12-core Intel Core i7 CPU, 32 GB DDR5-4800 memory, and an ASUS GeForce RTX 3080 Ti GPU (12 GB). The ECG annotation was performed using a custom-built ECG annotation tool (GitHub). AI-assisted tools were used only for code drafting and refactoring. Specifically, ChatGPT (OpenAI; during 2025, JST) and GitHub Copilot (GitHub; during 2025, JST) assisted in the Python boilerplate generation and readability improvements. All scripts were reviewed, tested, and finalized by the authors, who took full responsibility for the analyses. No confidential or patient-identifiable information was entered into the tool.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><sec id="s3-1"><title>Study Population and Baseline Characteristics</title><p>Between March 1, 2023, and November 20, 2025, 129 patients underwent monitoring with a single-lead garment-type wearable Holter ECG device. Of these, 12/129 (9.3%) patients with documented AT/AFL or paced rhythm were excluded according to the predefined task definition. Consequently, 117/129 (90.7%) patients were included in model development and internal validation. Their baseline characteristics are summarized in <xref ref-type="table" rid="table1">Table 1</xref>.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Baseline characteristics of the institutional cohort included in model development and internal validation at the University of Osaka Hospital (N=117).</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Characteristic</td><td align="left" valign="bottom">Value</td></tr></thead><tbody><tr><td align="left" valign="top">Age (y), mean (SD)</td><td align="left" valign="top">69.6 (12.5)</td></tr><tr><td align="left" valign="top">Sex (male), n (%)</td><td align="left" valign="top">77 (65.8)</td></tr><tr><td align="left" valign="top">Height (cm), mean (SD)</td><td align="left" valign="top">163.5 (9.4)</td></tr><tr><td align="left" valign="top">Weight (kg), mean (SD)</td><td align="left" valign="top">65.2 (14.9)</td></tr><tr><td align="left" valign="top">BMI (kg/m&#x00B2;), mean (SD)</td><td align="left" valign="top">24.2 (4.0)</td></tr><tr><td align="left" valign="top">History of AF<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup>, n (%)</td><td align="left" valign="top">99 (84.6)</td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>AF: atrial fibrillation.</p></fn></table-wrap-foot></table-wrap><p>The institutional dataset comprised a total of 2850 hours of ECG recordings from 117 patients. After segmentation and label aggregation, the number of unique window-level samples before data augmentation for the 1.5-, 3-, and 6-minute settings was 114,000, 56,969, and 28,452, respectively (<xref ref-type="fig" rid="figure4">Figure 4</xref>). Across all window length settings, non-AF windows were the most frequent class, whereas noise windows were the least frequent, indicating class imbalance in the generated datasets.</p><fig position="float" id="figure4"><label>Figure 4.</label><caption><p>Study cohort selection and dataset generation for model development and validation in garment-type wearable Holter ECG monitoring at the University of Osaka Hospital between March 1, 2023, and November 20, 2025. Flow diagram showing patient selection from the institutional cohort, exclusion of atrial tachycardia/atrial flutter (AT/AFL) and pacemaker cases according to the predefined task definition, and the number of generated R-R interval (RRI) histogram images for model development and internal validation. The number of generated RRI histogram images represents unique window-level samples before sliding window augmentation and corresponds to the total number of test windows across the 5 cross-validation folds. ECG: electrocardiogram.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig04.png"/></fig></sec><sec id="s3-2"><title>Rhythm and Signal Quality Characteristics in the Institutional Cohort</title><p>To characterize the rhythm composition of the institutional cohort, patient-level AF burden was calculated from the original 10-second annotations as the proportion of AF-labeled intervals among analyzable intervals. The AF burden distribution was bimodal: 53 (45.3%) patients had an AF burden of 0%, whereas 49 (41.9%) patients had an AF burden of &#x2265;80%; patients with intermediate AF burden were relatively infrequent (Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>).</p><p>In the rudimentary semiquantitative review of ectopic burden among 53 patients with an AF burden of 0%, estimated ectopic burden was &#x003C;1% in 36 (67.9%) patients, 1% to &#x003C;5% in 8 (15.1%) patients, 5% to &#x003C;10% in 3 (5.7%) patients, and &#x2265;10% in 6 (11.3%) patients. This review suggested that the non-AF recordings were not limited to exceptionally clean sinus rhythm. The distribution is summarized in Table S2 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><p>Noise-labeled intervals were commonly observed in the institutional recordings. The total number of noise-labeled 10-second intervals was 249,285, corresponding to approximately 692.5 hours of noise-labeled recording. At the patient level, the median noise burden was 17.6% (IQR 7%&#x2010;36%), and only 2 of 117 patients had a noise burden of 0%. These findings indicate that uninterpretable or unreliable segments were commonly present in long-term garment-type wearable Holter ECG recordings. The detailed distributions of noise burden are provided in Table S3 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s3-3"><title>Quality Assessment of R-Peak Detection</title><p>In the supplementary review of 300 randomly sampled 30-second ECG segments, 238 (79.3%) were considered annotatable. All pure non-AF segments and all pure AF segments were annotatable, and all these were graded as acceptable. Almost all pure noise segments (49/50, 98%) were judged nonannotatable and unreadable. Among noise-containing segments, 37 (74%) of 50 were annotatable and readable. Across the 238 annotatable segments, 237 (99.6%) were graded as acceptable, and 1 (0.4%) was graded as minor errors, whereas none were graded as unreliable.</p></sec><sec id="s3-4"><title>Annotation Agreement Assessment</title><p>In the supplementary annotation verification analysis, 150 randomly sampled ECG segments from the in-house dataset were reassessed, including 50 non-AF segments, 50 AF segments, and 50 noise segments. Agreement with the original study annotations was high for both the reevaluation by the original annotator and the independent review by an additional cardiologist. Cohen &#x03BA; was 0.91 for the comparison between the original annotations and the reevaluation by the original annotator and 0.95 for the comparison between the original annotations and the independent cardiologist, with corresponding agreement percentages of 94% and 96.7%, respectively. Detailed pairwise agreement (Cohen &#x03BA;) and class-stratified agreement by label are provided in Table S4 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s3-5"><title>Performance of the 2D-CNN Model in Internal Validation</title><p>The RRI histogram images were generated for each window length. Representative images of each class are shown in <xref ref-type="fig" rid="figure5">Figure 5</xref>. The proportion of candidate training windows excluded because they contained both AF and non-AF labels was less than 0.1% of all candidate windows. Exact ties requiring the application of the predefined tie-breaking rule were uncommon in the test data label aggregation process, occurring in 0% of 1.5-minute windows, 0.4% of 3-minute windows, and 0.3% of 6-minute windows. These low frequencies indicate that the tie-breaking rule had minimal influence on the overall evaluation. The mean performance metrics across the 5 cross-validation folds for each window length are presented in <xref ref-type="table" rid="table2">Table 2</xref>, and the detailed class-wise metrics, calculated in a one-versus-rest manner for the non-AF, AF, and noise classes, are provided in Table S5 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p><fig position="float" id="figure5"><label>Figure 5.</label><caption><p>Generation of R-R interval (RRI) histogram images and representative examples used for 3-class classification. Schematic of the transformation from the RRI time series to 2D histogram images, together with representative examples of non&#x2013;atrial fibrillation (AF), AF, and noise windows. The noise example represents an actual artifact-contaminated waveform recorded using the garment-type wearable Holter ECG device and the corresponding RRI-derived heart rate histogram, not simulated Gaussian noise.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig05.png"/></fig><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Window-level performance metrics across the 5 cross-validation folds for the 1.5-, 3-, and 6-min models in the internal validation cohort.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Window</td><td align="left" valign="bottom">1.5 min, mean (SD)</td><td align="left" valign="bottom">3 min, mean (SD)</td><td align="left" valign="bottom">6 min, mean (SD)</td></tr></thead><tbody><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">0.9656 (0.0116)</td><td align="left" valign="top">0.9664 (0.0110)</td><td align="left" valign="top">0.9610 (0.0106)</td></tr><tr><td align="left" valign="top">Macro <italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">0.9633 (0.0115)</td><td align="left" valign="top">0.9641 (0.0109)</td><td align="left" valign="top">0.9582 (0.0120)</td></tr><tr><td align="left" valign="top">Macro AUROC<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup><sup>,</sup><sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup></td><td align="left" valign="top">0.9958 (0.0026)</td><td align="left" valign="top">0.9961 (0.0025)</td><td align="left" valign="top">0.9955 (0.0022)</td></tr><tr><td align="left" valign="top">AF<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>_Sensitivity</td><td align="left" valign="top">0.9675 (0.0318)</td><td align="left" valign="top">0.9666 (0.0268)</td><td align="left" valign="top">0.9489 (0.0418)</td></tr><tr><td align="left" valign="top">AF_Specificity</td><td align="left" valign="top">0.9800 (0.0090)</td><td align="left" valign="top">0.9817 (0.0081)</td><td align="left" valign="top">0.9835 (0.0037)</td></tr><tr><td align="left" valign="top">AF_PPV<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td><td align="left" valign="top">0.9558 (0.0220)</td><td align="left" valign="top">0.9591 (0.0214)</td><td align="left" valign="top">0.9636 (0.0109)</td></tr><tr><td align="left" valign="top">AF_NPV<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">0.9859 (0.0135)</td><td align="left" valign="top">0.9849 (0.0112)</td><td align="left" valign="top">0.9761 (0.0187)</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>Calculated using a one-versus-rest approach across the 3 classes.</p></fn><fn id="table2fn2"><p><sup>b</sup>AUROC: area under the receiver operating characteristic curve.</p></fn><fn id="table2fn3"><p><sup>c</sup>AF: atrial fibrillation.</p></fn><fn id="table2fn4"><p><sup>d</sup>PPV: positive predictive value.</p></fn><fn id="table2fn5"><p><sup>e</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap><p>The 1.5- and 3-minute windows achieved similarly high overall performance, with mean (SD) cross-validation accuracies of 96.6% (SD 1.2) for both models, macro <italic>F</italic><sub>1</sub>-scores of 96.3% (SD 1.2) and 96.4% (SD 1.1), and macro AUROCs of 99.6% (SD 0.3) and 99.6% (SD 0.3), respectively. For the AF class, the 3-minute model showed slightly better balance than the 1.5-minute model, with comparable sensitivity of 96.7% (SD 2.7) versus 96.7% (SD 3.2), and slightly higher specificity of 98.2% (SD 0.8) versus 98.0% (SD 0.9) and PPV of 95.9% (SD 2.1) versus 95.6% (SD 2.2). In contrast, the 6-minute model showed slightly lower overall accuracy of 96.1% (SD 1.1), macro F1-score of 95.8% (SD 1.2), and AF sensitivity of 94.9% (SD 4.2), although AF specificity remained high at 98.4% (SD 0.4). Performance for the non-AF class was also consistently high across window lengths, indicating stable discrimination of AF from non-AF rhythms in the 3-class framework. Noise class performance appeared favorable in the internal validation, although its interpretation should remain cautious in light of the lower external noise sensitivity discussed in the following paragraph. Receiver operating characteristic curves for each window length are shown in Figure S3 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>In an exploratory post hoc subgroup analysis by garment type using pooled out-of-fold predictions from the 3-minute model, performance was similar between shirt-type and belt-type recordings. Accuracy was 96.8% in shirt-type recordings and 96.5% in belt-type recordings, and AF sensitivity was 97.4 and 96.7, respectively. However, the shirt-type subgroup included fewer patients and showed substantial class imbalance; therefore, definitive conclusions regarding garment-specific performance differences cannot be drawn. The detailed subgroup results are provided in Table S6 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>.</p></sec><sec id="s3-6"><title>Performance of the 2D-CNN Model in External Validation</title><p>The final DL models, trained using the 117-patient institutional dataset, were evaluated using the MIT-BIH AFDB to assess their generalizability. In the primary external validation, AFL-annotated intervals were excluded to align the test-label definition with the training task, in which AT/AFL episodes had been excluded. The results are summarized in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Window-level performance metrics in the primary external validation analysis using the MIT-BIH<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> Atrial Fibrillation Database (AFDB).</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">MIT-BIH AFDB</td><td align="left" valign="bottom">1.5-min window model</td><td align="left" valign="bottom">3-min window model</td><td align="left" valign="bottom">6-min window model</td></tr></thead><tbody><tr><td align="left" valign="top">Accuracy</td><td align="left" valign="top">0.9659</td><td align="left" valign="top">0.9731</td><td align="left" valign="top">0.9663</td></tr><tr><td align="left" valign="top">Non-AF<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup>_Sensitivity</td><td align="left" valign="top">0.9836</td><td align="left" valign="top">0.9774</td><td align="left" valign="top">0.9805</td></tr><tr><td align="left" valign="top">AF_Sensitivity</td><td align="left" valign="top">0.9416</td><td align="left" valign="top">0.9695</td><td align="left" valign="top">0.9469</td></tr><tr><td align="left" valign="top">Noise_Sensitivity</td><td align="left" valign="top">0.7959</td><td align="left" valign="top">0.7000</td><td align="left" valign="top">0.8000</td></tr><tr><td align="left" valign="top">Non-AF_Specificity</td><td align="left" valign="top">0.9448</td><td align="left" valign="top">0.9676</td><td align="left" valign="top">0.9464</td></tr><tr><td align="left" valign="top">AF_Specificity</td><td align="left" valign="top">0.9864</td><td align="left" valign="top">0.9772</td><td align="left" valign="top">0.9814</td></tr><tr><td align="left" valign="top">Noise_Specificity</td><td align="left" valign="top">0.9963</td><td align="left" valign="top">0.9998</td><td align="left" valign="top">0.9991</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>MIT-BIH: Massachusetts Institute of Technology&#x2013;Beth Israel Hospital</p></fn><fn id="table3fn2"><p><sup>b</sup>AF: atrial fibrillation.</p></fn></table-wrap-foot></table-wrap><p>The external validation dataset consisted of 23 independent AFDB recordings (approximately 230 h in total). As the AFDB did not originally include an explicit noise label, visually uninterpretable windows were additionally annotated as noise. The resulting reference-label distribution was imbalanced across all window length settings: for the 1.5-minute setting, 5549 non-AF (59.7%), 3701 AF (39.8%), and 49 noise (0.5%) windows; for the 3-minute setting, 2785 non-AF (60%), 1834 AF (39.5%), and 20 noise (0.5%) windows; and for the 6-minute setting, 1385 non-AF (59.8%), 922 AF (39.8%), and 10 noise (0.4%) windows.</p><p>In the AFDB, the 3-minute model achieved the window-level accuracy of 97.3%, with an AF sensitivity of 96.9% and AF specificity of 97.7%. The 1.5-minute model achieved an overall accuracy of 96.6%, AF sensitivity of 94.2%, and AF specificity of 98.6%, representing a modest reduction in overall accuracy and AF sensitivity compared to the 3-minute model. The 6-minute model showed overall accuracy of 96.6%, AF sensitivity of 94.7%, and AF specificity of 98.1%, which were slightly lower than those of the 3-minute window. Accordingly, in the external dataset, the 3-minute window showed the highest overall accuracy and AF sensitivity while maintaining high AF specificity, although the differences across window lengths were modest and no formal statistical comparison was performed. Across all window lengths, however, sensitivity for the noise class was low: 79.6% (39/49), 70% (14/20), and 80% (8/10) for the 1.5-, 3-, and 6-minute models, respectively. This result should be interpreted cautiously because the number of noise-labeled windows in the external dataset was small and the noise characteristics in the AFDB may not have fully matched those in the institutional wearable dataset. Consistent with the class-wise metrics, the confusion matrices showed that AF and non-AF were generally well discriminated, whereas noise windows were relatively infrequent and more often misclassified than the other classes (<xref ref-type="fig" rid="figure6">Figure 6</xref>).</p><fig position="float" id="figure6"><label>Figure 6.</label><caption><p>Confusion matrices for external validation of the 3-class classification task in the Massachusetts Institute of Technology&#x2013;Beth Israel Hospital (MIT-BIH) Atrial Fibrillation Database (AFDB). Confusion matrices for the primary external validation analysis using 23 independent MIT-BIH AFDB recordings, with atrial flutter&#x2013;annotated intervals excluded to match the institutional training task. The x-axis indicates reference labels, and the y-axis indicates model predictions for non&#x2013;atrial fibrillation (AF), AF, and noise across the evaluated window lengths.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig06.png"/></fig><p>To further characterize the low noise sensitivity, we reviewed the false-negative noise windows in the 3-minute external validation analysis. Among the 20 noise-labeled windows, 6 were misclassified as non-AF or AF. Visual review of the corresponding ECG waveforms and RRI histograms suggested 2 common patterns. In 3 windows, the ECG waveform was visually nondiagnostic for rhythm interpretation, but the automatic R-peak detection algorithm generated structured RRI sequences, resulting in RRI histograms that did not show a typical scattered noise pattern. In the remaining 3 windows, visually noisy ECG portions were mixed with analyzable rhythm portions within the same 3-minute window, leaving structured RRI features in the model input despite the window-level noise label. Representative examples are shown in Figure S4 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>.</p><p>We visually inspected the windows misclassified by the 3-minute model to characterize common patterns in the corresponding images. Among the AF windows misclassified as non-AF, 87.5% (49/56) corresponded to images in which AF and non-AF rhythms coexisted within the same 3-minute window, whereas windows in which AF clearly persisted beyond 3 minutes but was not detected accounted for 12.5% (7/56) of the cases. These findings indicate that most AF-to-non-AF misclassifications were associated with mixed-rhythm windows. Representative misclassified images are shown in <xref ref-type="fig" rid="figure7">Figure 7</xref>.</p><fig position="float" id="figure7"><label>Figure 7.</label><caption><p>Representative misclassified atrial fibrillation (AF) windows in the primary external validation analysis. Representative examples from the 3-min model showing AF windows misclassified as non-AF in the MIT-BIH Atrial Fibrillation Database. The figure highlights cases in which AF and non-AF rhythms coexisted within a single window, as well as cases in which AF persisted beyond 3 min but was not detected.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig07.png"/></fig><p>The results of the secondary analysis in which AFL intervals were mapped to the AF class are provided in Table S7 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>. Across all window lengths, performance in this secondary analysis was slightly lower than in the primary AFL-excluded analysis; for example, for the 3-minute model, accuracy decreased from 97.3% to 96.1%, and AF sensitivity decreased from 96.9% to 93.3%.</p></sec><sec id="s3-7"><title>Evaluation of AF Burden</title><p>When AF burden was treated as a continuous patient-level variable derived from aggregated window-level classifications, the model showed a strong correlation with the reference AF burden across all 3 window length settings (<xref ref-type="fig" rid="figure8">Figure 8</xref>). Pearson correlation coefficients were 0.995 (<italic>R</italic>&#x00B2;=0.991), 0.991 (<italic>R</italic>&#x00B2;=0.982), and 0.989 (<italic>R</italic>&#x00B2;=0.979) for the 1.5-, 3-, and 6-minute models, respectively, and the corresponding Spearman rank correlation coefficients were 0.988, 0.982, and 0.979. Overall, patient-level AF burden correlation was high across all window lengths, with only small differences between models.</p><fig position="float" id="figure8"><label>Figure 8.</label><caption><p>Patient-level atrial fibrillation (AF) burden correlation in the MIT-BIH Atrial Fibrillation Database external validation cohort. Scatter plot and regression line comparing reference AF burden with model-estimated AF burden for each of the 23 external validation recordings included in the primary atrial flutter&#x2013;excluded analysis. AF burden was evaluated at the patient level by aggregating reference intervals and model predictions across each recording. RSE: residual standard error.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="medinform_v14i1e91960_fig08.png"/></fig></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><sec id="s4-1"><title>Principal Findings</title><p>In this study, we developed a DL model that classifies RRI-derived 2D histogram images into non-AF, AF, and noise classes using a 2D-CNN. Internal and external validation suggested that the 3-minute window provided a clinically reasonable balance across the evaluated window lengths, although the differences were modest and should be interpreted as exploratory rather than definitive. In the external validation, the model also showed high correlation with the reference AF burden at the patient level. The contribution of this study lies not in the use of a novel network architecture, but in the application and evaluation of an RRI-based 3-class classification framework that explicitly incorporates noise in real-world garment-type wearable Holter ECG recordings.</p></sec><sec id="s4-2"><title>Comparison to Prior Work</title><p>For automated AF diagnosis, DL approaches can be broadly grouped into waveform-based models, RRI/heart rate variability (HRV)&#x2013;based models, and multimodal frameworks that combine ECG-derived features with additional clinical information. Waveform-based models trained directly on raw ECG signals have shown high diagnostic performance in a variety of datasets, including residual network&#x2013;based approaches, but their performance may depend substantially on recording format, window length, and dataset characteristics [<xref ref-type="bibr" rid="ref18">18</xref>-<xref ref-type="bibr" rid="ref22">22</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. In contrast, RRI/HRV-based approaches focus on the beat-to-beat irregularity that is fundamental to AF and have been explored using transition-matrix features, entropy-based indices, multivariate classifiers, visual irregularity mapping, Lorenz/Poincar&#x00E9;-type representations, and image-based DL frameworks [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>,<xref ref-type="bibr" rid="ref32">32</xref>-<xref ref-type="bibr" rid="ref34">34</xref>]. Recent multimodal work has further suggested that combining ECG with HRV and demographic information may improve AF detection performance in selected settings [<xref ref-type="bibr" rid="ref35">35</xref>]. These studies provide useful context for the present findings, but direct numerical comparison should be interpreted cautiously because the underlying datasets, recording durations, input modalities, and intended use cases differ substantially. In this context, this study does not seek novelty in the network architecture itself; rather, its contribution lies in evaluating an RRI-based, 3-class, noise-aware framework in long-term garment-type wearable Holter ECG recordings, a setting in which noisy and clinically heterogeneous data are common. By converting the RRI time series into histogram images, this approach preserves AF-related irregularity patterns in a form suitable for CNN-based analysis while remaining conceptually closer to established HRV-based rhythm assessment than to raw waveform classification. As RRI-based representations use interval-derived information rather than full raw waveform input, converting an RRI time series into 2D images for CNN-based analysis may provide a conceptually simple and practical framework for AF screening in noisy wearable ECG recordings.</p><p>With a single-lead Holter or wearable ECG, reliably identifying P waves and fibrillatory waves (f waves), which are important for conventional AF diagnosis, can be challenging in the presence of noise. This problem is particularly pronounced in garment-type devices (eg, shirt- or belt-type systems), where artifacts arising from body movements and variable electrode contacts are common [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref36">36</xref>]. Garment-type Holter systems are expected to reduce the skin burden associated with adhesive electrodes, but they may be associated with more frequent noise contamination because of motion artifacts and variable electrode contact. Compared to curated public databases, real-world long-term Holter recordings contain a greater proportion of segments affected by noise artifacts. If such noise is not adequately identified, model performance, particularly specificity, may deteriorate, leading to an increased number of false positives [<xref ref-type="bibr" rid="ref37">37</xref>]. RRI-based analysis offers a practical advantage in this setting; provided that R-peaks are accurately detected, it may be less dependent on raw waveform morphology than waveform-based approaches, making it attractive for handling noisy wearable ECG recordings [<xref ref-type="bibr" rid="ref38">38</xref>].</p><p>To better handle noise-contaminated wearable ECG data, we implemented two additional strategies: (1) histogram binning of the RRI time series and (2) explicit annotation and learning of noise as a separate class. Histogram binning may provide a degree of local aggregation, thereby reducing the effects of minor timing fluctuations or isolated R-peak detection errors [<xref ref-type="bibr" rid="ref39">39</xref>]. In addition, by formulating the task as a 3-class problem (non-AF, AF, and noise), the model was designed to distinguish uninterpretable segments from AF and non-AF rhythms without requiring a separate preprocessing step to exclude noisy data in advance. This framework may be advantageous for real-world wearable Holter ECG recordings, in which noise contamination is often unavoidable. Several previous studies have reported higher classification accuracies after rigorously excluding noisy segments from analysis [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref34">34</xref>,<xref ref-type="bibr" rid="ref40">40</xref>]. However, such approaches may be less applicable to daily wearable monitoring, where preprocessing and manual cleaning are not always feasible. In this study, the AF diagnostic performance remained high despite the inclusion of noise-containing recordings, suggesting that this 3-class noise-aware framework is practical for AF screening in garment-type wearable ECG monitoring.</p><p>At the same time, noise sensitivity was substantially lower in the external validation than in the internal validation. One possible explanation is the very small proportion of noise-labeled segments in the external dataset, which makes sensitivity estimates unstable. Another possibility is a domain shift in noise characteristics: many noise segments in the institutional wearable dataset were associated with relatively prolonged signal degradation, such as electrode detachment or sustained signal loss, whereas the noise segments additionally annotated in the AFDB were retrospectively identified as visually apparent artifacts in a public dataset that did not originally include an explicit noise label. Additional review of false-negative noise windows suggested that some visually nondiagnostic ECG segments still yielded structured RRI histograms after automatic R-peak detection, while others contained mixed noisy and analyzable ECG portions within the same window. Thus, waveform-level visual noise annotation did not always correspond to degradation of the RRI-based feature representation used by the model, limiting the generalizability of noise classification across datasets. Although lower noise sensitivity should be regarded as a limitation, the consistently high noise specificity indicates that the model rarely assigned analyzable AF or non-AF windows to the noise class. This may be relevant in clinical review workflows, where excessive exclusion of interpretable rhythm segments could also be undesirable. The noise class in the present framework should therefore be interpreted as an operational class to avoid forcing unreliable windows into AF or non-AF categories, rather than as a universal artifact detector. Future implementations may benefit from complementary signal quality assessment, wear status detection, or preprocessing procedures.</p><p>The choice of the analysis window length for AF detection must be aligned with the model performance and the clinical significance of AF episodes. Previous studies using cardiac implantable electronic devices have suggested that stroke risk may increase when atrial high-rate episodes persist for &#x2265;6 minutes, and 6 minutes have therefore been proposed as a clinically relevant threshold [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>]. In this study, considering this evidence, we evaluated window lengths of 1.5, 3, and 6 minutes. Both internal and external validations suggested that the 3-minute window provided a clinically reasonable balance between AF sensitivity and specificity. However, the differences across window lengths were modest, and no formal statistical comparison was performed; therefore, this interpretation should be regarded as exploratory rather than definitive. The 3-minute model showed a high patient-level correlation with the reference AF burden in the AFDB, with performance comparable to that reported in previous studies [<xref ref-type="bibr" rid="ref4">4</xref>,<xref ref-type="bibr" rid="ref41">41</xref>]. A plausible explanation for the slightly lower AF sensitivity of the 6-minute model is that longer windows are more likely to include mixed rhythm or transitional segments, which may dilute AF-related irregularity within a single window; however, the high patient-level AF burden correlation suggests that the practical impact of this effect on overall burden estimation was limited. These window-level classification metrics and patient-level AF burden estimates address complementary aspects of performance: the former reflects the model&#x2019;s ability to classify individual ECG windows, whereas the latter reflects the clinical interpretability of aggregating those predictions over an entire recording. Although small differences in AF burden estimates may not necessarily alter patient management in every clinical context, this analysis was included to show that the aggregated model outputs remained clinically meaningful beyond segment-level classification accuracy alone. From a practical monitoring perspective, a 3-minute window also allows AF status to be updated at relatively short intervals while still permitting consecutive-window aggregation to identify clinically relevant sustained episodes. Collectively, these findings indicate that a 3-minute RRI-based model may provide a practical balance for AF screening and AF burden assessment in ambulatory long-term Holter monitoring while allowing consecutive-window aggregation to support the detection of sustained AF episodes.</p></sec><sec id="s4-3"><title>Future Directions</title><p>As the present model was specifically designed to detect AF on the basis of RRI irregularity, AT/AFL episodes were excluded from the primary training and validation framework. AT/AFL often exhibit relatively regular RRIs and therefore do not consistently share the irregular rhythm pattern that characterizes AF, making them difficult to detect reliably using an RRI irregularity&#x2013;based approach alone [<xref ref-type="bibr" rid="ref26">26</xref>]. In a secondary analysis, in which AFL was grouped with the AF class for broader clinical categorization, diagnostic performance decreased, further supporting the limited suitability of the present RRI-based framework for identifying AT/AFL as AF. Therefore, future studies should explore a 2-stage algorithm in which RRI-based screening is followed by a detailed waveform-based analysis to identify AT/AFL among segments classified as non-AF. Incorporating additional features, such as waveform morphology and periodicity measures, may help construct a more comprehensive framework for diagnosing supraventricular arrhythmias using wearable ECG devices [<xref ref-type="bibr" rid="ref42">42</xref>].</p><p>As the demand for AI-based diagnosis from single-lead ECG data increases, AF detection algorithms embedded in standard smartwatches still face limitations, including &#x201C;unclassifiable&#x201D; recordings and suboptimal sensitivity and specificity [<xref ref-type="bibr" rid="ref43">43</xref>,<xref ref-type="bibr" rid="ref44">44</xref>]. Conversely, the direct application of DL to smartwatch ECG waveforms can yield a performance that markedly surpasses that of built-in algorithms, suggesting that the potential for AI applications in wearable ECG will continue to grow. Within this context, the RRI-based 2D-CNN model developed in this study appears to be a promising AF screening approach that combines noise-aware design with potential practical applicability in prolonged high noise settings, such as garment-type wearable Holter monitoring. Our findings may contribute to the future implementation of garment-type wearable diagnostic systems and the development of efficient AF screening tools in primary care and community-based health care settings.</p></sec><sec id="s4-4"><title>Limitations</title><p>Although this study demonstrated the effectiveness of an RRI-based automatic AF detection model using single-lead Holter ECG recordings, it has several limitations.</p><p>First, although the CNN-based DL model achieved high classification performance, its decision-making process relied on internal latent representations and was not directly interpretable at the individual-prediction level. This inherent &#x201C;black-box&#x201D; nature reflects the broader challenge of explainability in medical AI and underscores the need for complementary approaches, such as model-agnostic explanation techniques, if the system is to be used for clinical decision support.</p><p>Second, because AT/AFL episodes and paced rhythms were excluded from the primary training and validation framework according to the predefined task definition, the present model was not designed or evaluated as a general classifier of supraventricular tachyarrhythmias or paced ECG recordings. In particular, pacemakers may artificially regulate RRIs and thereby alter the RRI irregularity patterns on which the present algorithm depends. Accordingly, its applicability to AT/AFL detection and to recordings containing paced rhythms remains limited and should be addressed in future studies using dedicated datasets and complementary features beyond RRI irregularity alone.</p><p>Third, although internal validation was performed using institutional data and external validation was conducted using the AFDB, the model&#x2019;s generalizability across different clinical settings, patient populations, recording conditions, and ECG devices remains uncertain. In particular, the institutional and external validation datasets were enriched for AF relative to a general screening population. As PPV and NPV are mathematically dependent on disease prevalence, the favorable predictive values observed in this study may not directly translate to lower-prevalence screening settings; in particular, the PPV may be lower when the algorithm is applied to populations in which the true prevalence of AF is low. Noise characteristics can also vary substantially among wearable devices and datasets, and this may partly explain the markedly lower noise sensitivity observed in the external validation. Future studies should determine whether the same model, or a suitably adapted version, can be applied across diverse hardware platforms, external noise conditions, and lower-prevalence screening populations. In addition, because windows containing both AF and non-AF labels were excluded from training to avoid ambiguous supervisory signals, the reported classification performance may be somewhat optimistic relative to more heterogeneous real-world recordings containing frequent rhythm transitions. The proportion of such excluded candidate training windows was less than 0.1% in the present dataset. Consistent with this low transition rate, the patient-level AF burden distribution was bimodal, with many patients showing either no AF or very high AF burden and relatively few patients showing intermediate AF burden. Therefore, patients with highly active PAF characterized by frequent transitions between AF and non-AF rhythms were likely underrepresented. Moreover, because window-level reference labels were assigned using a majority vote rule, short AF episodes occupying less than half of a given window could be underrepresented, particularly in longer windows such as the 6-minute setting. Thus, the present framework should be interpreted as a window-level AF screening and burden estimation approach rather than a definitive episode-level detector for very brief AF events.</p><p>Fourth, the proposed method depends on automatic R-peak detection because the model input was derived from RRI data. Potential sources of R-peak detection error include noise and artifacts in Holter or wearable ECG recordings, as well as limitations in the acquisition system. Although the device incorporated analog band limiting intended for antialiasing before 250 Hz ADC sampling, the available specifications did not indicate a sharp low-pass filter below 100 Hz; therefore, residual aliasing of high-frequency components cannot be completely excluded. To assess the practical reliability of the R-peak detection step, we performed a supplementary manual quality assessment using 300 randomly sampled 30-second ECG segments; however, this was not an exhaustive beat-level validation of all 2850 hours of institutional recordings. In the external validation, manually annotated noise segments in the AFDB may not have fully reproduced the sustained mechanical artifacts encountered in garment-type wearable ECG recordings, and waveform-level visual noise labels may not always correspond to degradation of RRI-based feature representations. These factors may have contributed to the low noise sensitivity observed in the external validation. In addition, because the RRI-derived histogram was limited to a predefined heart rate range of 30 to 400 beats per minute, RRIs longer than 2.0 seconds were not explicitly represented as prolonged-pause features. Thus, clinically significant pauses, such as those observed in tachy-brady syndrome, would require separate pause-detection logic or waveform-based review.</p><p>Fifth, although we performed a rudimentary semiquantitative review of ectopic burden in patients with 0% AF burden, this assessment was based on sampled non-AF ECG segments and was not a formal beat-by-beat adjudication of premature atrial contractions or premature ventricular contractions across the entire non-AF class. The review suggested that the non-AF recordings were not limited to exceptionally clean sinus rhythm, but the precise prevalence and distribution of ectopic beats in all non-AF windows remain unknown. As the present model relies entirely on RRI irregularity, frequent ectopy may mimic AF-like irregularity, and may affect the interpretation of the model&#x2019;s true specificity.</p></sec><sec id="s4-5"><title>Conclusions</title><p>This study developed and evaluated an RRI-based AF screening approach in which RRI time series data extracted from single-lead garment-type Holter ECGs were transformed into 2D images for DL-based classification. A 2D-CNN model trained on these visualized RRI features demonstrated high classification accuracy in both internal and external validations, suggesting potential utility for AF screening in prolonged monitoring settings using recordings without paced rhythms; recordings containing paced rhythms should currently be considered outside the intended use of this RRI-based algorithm. The 3-class framework, which explicitly separated noise from AF and non-AF rhythms, may be useful for handling real-world garment-type recordings that contain uninterpretable segments. However, broader validation across more diverse patient populations, recording environments, and device types will be necessary before routine clinical deployment can be considered. Importantly, the real-world performance of this framework will depend on the reliability of R-peak detection and on device- and dataset-specific noise characteristics.</p><p>Future work should further evaluate real-time feasibility and explore integration with waveform-based approaches for other supraventricular arrhythmias, such as AT/AFL.</p></sec></sec></body><back><ack><p>ChatGPT (OpenAI) was used to improve the grammar, clarity, and style of the author-written English text and to assist with limited English translation; and DeepL was used for translation from Japanese to English. These tools were not used to generate original scientific content, analyses, or references. All outputs were verified and edited by the authors, who accept full responsibility for the final manuscript. No confidential or patient-identifiable data were provided to these services.</p></ack><notes><sec><title>Funding</title><p>This study was conducted under a collaborative research agreement with the Mitsufuji Corporation, which provided research funding to the University of Osaka.</p></sec><sec><title>Data Availability</title><p>The electrocardiographic (ECG) data used in this study were obtained from patients who provided informed consent. The study protocol was approved by the Institutional Ethics Committee of the University Hospital. Due to ethical and legal restrictions related to patient privacy, raw ECG data are not publicly available. Anonymized and processed data (including derived R-R interval features and model input representations), as well as the analysis code, are available from the corresponding author upon reasonable request for research purposes and are subject to institutional approval. The in-house ECG annotation tool is available on GitHub [<xref ref-type="bibr" rid="ref45">45</xref>].</p></sec></notes><fn-group><fn fn-type="conflict"><p>TN, KO, TO, and YS reported research funding (to their institutions) from Mitsufuji Corporation related to this work. The other authors declare no competing interests.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">1D-CNN</term><def><p>one-dimensional convolutional neural network</p></def></def-item><def-item><term id="abb2">2D-CNN</term><def><p>two-dimensional convolutional neural network</p></def></def-item><def-item><term id="abb3">ADC</term><def><p>analog-to-digital converter</p></def></def-item><def-item><term id="abb4">AF</term><def><p>atrial fibrillation</p></def></def-item><def-item><term id="abb5">AFDB</term><def><p>MIT-BIH Atrial Fibrillation Database</p></def></def-item><def-item><term id="abb6">AFL</term><def><p>atrial flutter</p></def></def-item><def-item><term id="abb7">AT</term><def><p>atrial tachycardia</p></def></def-item><def-item><term id="abb8">AUROC</term><def><p>area under the receiver operating characteristic curve</p></def></def-item><def-item><term id="abb9">DL</term><def><p>deep learning</p></def></def-item><def-item><term id="abb10">ECG</term><def><p>electrocardiography</p></def></def-item><def-item><term id="abb11">HRV</term><def><p>heart rate variability</p></def></def-item><def-item><term id="abb12">JR</term><def><p>junctional rhythm</p></def></def-item><def-item><term id="abb13">MIT-BIH</term><def><p>Massachusetts Institute of Technology - Beth Israel Hospital</p></def></def-item><def-item><term id="abb14">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb15">NR</term><def><p>normal rhythm</p></def></def-item><def-item><term id="abb16">PAF</term><def><p>paroxysmal atrial fibrillation</p></def></def-item><def-item><term id="abb17">PPV</term><def><p>positive predictive value</p></def></def-item><def-item><term id="abb18">RRI</term><def><p>R-R interval</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><collab>Writing Committee Members</collab><name name-style="western"><surname>Joglar</surname><given-names>JA</given-names> </name><name name-style="western"><surname>Chung</surname><given-names>MK</given-names> </name><etal/></person-group><article-title>2023 ACC/AHA/ACCP/HRS Guideline for the Diagnosis and Management of Atrial Fibrillation: a report of the American College of Cardiology/American Heart Association Joint Committee on Clinical Practice Guidelines</article-title><source>J Am Coll Cardiol</source><year>2024</year><month>01</month><day>2</day><volume>83</volume><issue>1</issue><fpage>109</fpage><lpage>279</lpage><pub-id pub-id-type="doi">10.1016/j.jacc.2023.08.017</pub-id><pub-id pub-id-type="medline">38043043</pub-id></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Nogami</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kurita</surname><given-names>T</given-names> </name><name name-style="western"><surname>Kusano</surname><given-names>K</given-names> </name><etal/></person-group><article-title>JCS/JHRS 2021 guideline focused update on non-pharmacotherapy of cardiac arrhythmias</article-title><source>J Arrhythm</source><year>2022</year><month>02</month><volume>38</volume><issue>1</issue><fpage>1</fpage><lpage>30</lpage><pub-id pub-id-type="doi">10.1002/joa3.12649</pub-id><pub-id pub-id-type="medline">35222748</pub-id></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fiorina</surname><given-names>L</given-names> </name><name name-style="western"><surname>Maupain</surname><given-names>C</given-names> </name><name name-style="western"><surname>Gardella</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Evaluation of an ambulatory ECG analysis platform using deep neural networks in routine clinical practice</article-title><source>J Am Heart Assoc</source><year>2022</year><month>09</month><day>20</day><volume>11</volume><issue>18</issue><fpage>e026196</fpage><pub-id pub-id-type="doi">10.1161/JAHA.122.026196</pub-id><pub-id pub-id-type="medline">36073638</pub-id></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhang</surname><given-names>P</given-names> </name><name name-style="western"><surname>Lin</surname><given-names>F</given-names> </name><name name-style="western"><surname>Ma</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Automatic screening of patients with atrial fibrillation from 24-h Holter recording using deep learning</article-title><source>Eur Heart J Digit Health</source><year>2023</year><month>05</month><volume>4</volume><issue>3</issue><fpage>216</fpage><lpage>224</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztad018</pub-id><pub-id pub-id-type="medline">37265871</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Naydenov</surname><given-names>S</given-names> </name><name name-style="western"><surname>Jekova</surname><given-names>I</given-names> </name><name name-style="western"><surname>Krasteva</surname><given-names>V</given-names> </name></person-group><article-title>Recognition of supraventricular arrhythmias in Holter ECG recordings by ECHOView color map: a case series study</article-title><source>J Cardiovasc Dev Dis</source><year>2023</year><month>08</month><day>24</day><volume>10</volume><issue>9</issue><fpage>360</fpage><pub-id pub-id-type="doi">10.3390/jcdd10090360</pub-id><pub-id pub-id-type="medline">37754789</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rahman</surname><given-names>MM</given-names> </name><name name-style="western"><surname>Rivolta</surname><given-names>MW</given-names> </name><name name-style="western"><surname>Badilini</surname><given-names>F</given-names> </name><name name-style="western"><surname>Sassi</surname><given-names>R</given-names> </name></person-group><article-title>Uncertainty estimation of deep learning models for atrial fibrillation detection from Holter recordings: a benchmark study</article-title><source>Biomed Signal Process Control</source><year>2026</year><month>03</month><volume>113</volume><fpage>109032</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2025.109032</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shrikanth Rao</surname><given-names>SK</given-names> </name><name name-style="western"><surname>Kolekar</surname><given-names>MH</given-names> </name><name name-style="western"><surname>Martis</surname><given-names>RJ</given-names> </name></person-group><article-title>Atrial fibrillation detection using Poincare geometry and heart beat intervals</article-title><source>Expert Syst</source><year>2023</year><month>08</month><volume>40</volume><issue>7</issue><fpage>e13277</fpage><pub-id pub-id-type="doi">10.1111/exsy.13277</pub-id></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Machino</surname><given-names>T</given-names> </name><name name-style="western"><surname>Aonuma</surname><given-names>K</given-names> </name><name name-style="western"><surname>Maruo</surname><given-names>K</given-names> </name><etal/></person-group><article-title>Randomized crossover trial of 2-week Garment electrocardiogram with dry textile electrode to reveal instances of post-ablation recurrence of atrial fibrillation underdiagnosed during 24-hour Holter monitoring</article-title><source>PLOS ONE</source><year>2023</year><volume>18</volume><issue>2</issue><fpage>e0281818</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0281818</pub-id><pub-id pub-id-type="medline">36827294</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Joseph Michael Jerard</surname><given-names>V</given-names> </name><name name-style="western"><surname>Thilagaraj</surname><given-names>M</given-names> </name><name name-style="western"><surname>Pandiaraj</surname><given-names>K</given-names> </name><name name-style="western"><surname>Easwaran</surname><given-names>M</given-names> </name><name name-style="western"><surname>Govindan</surname><given-names>P</given-names> </name><name name-style="western"><surname>Elamaran</surname><given-names>V</given-names> </name></person-group><person-group person-group-type="editor"><name name-style="western"><surname>B</surname><given-names>MA</given-names> </name></person-group><article-title>Reconfigurable architectures with high-frequency noise suppression for wearable ECG devices</article-title><source>J Healthc Eng</source><year>2021</year><volume>2021</volume><pub-id pub-id-type="pmid">1552641</pub-id><pub-id pub-id-type="doi">10.1155/2021/1552641</pub-id><pub-id pub-id-type="medline">34976322</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Amami</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yoshihisa</surname><given-names>A</given-names> </name><name name-style="western"><surname>Horikoshi</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Utility of a novel wearable electrode embedded in an undershirt for electrocardiogram monitoring and detection of arrhythmias</article-title><source>PLOS ONE</source><year>2022</year><volume>17</volume><issue>8</issue><fpage>e0273541</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0273541</pub-id><pub-id pub-id-type="medline">35998187</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>McKenna</surname><given-names>S</given-names> </name><name name-style="western"><surname>McCord</surname><given-names>N</given-names> </name><name name-style="western"><surname>Diven</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Evaluating the impacts of digital ECG denoising on the interpretive capabilities of healthcare professionals</article-title><source>Eur Heart J Digit Health</source><year>2024</year><month>09</month><volume>5</volume><issue>5</issue><fpage>601</fpage><lpage>610</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztae063</pub-id><pub-id pub-id-type="medline">39318698</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aminorroaya</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dhingra</surname><given-names>LS</given-names> </name><name name-style="western"><surname>Pedroso</surname><given-names>AF</given-names> </name><etal/></person-group><article-title>Development and multinational validation of an ensemble deep learning algorithm for detecting and predicting structural heart disease using noisy single-lead electrocardiograms</article-title><source>Eur Heart J Digit Health</source><year>2025</year><month>07</month><volume>6</volume><issue>4</issue><fpage>554</fpage><lpage>566</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztaf034</pub-id><pub-id pub-id-type="medline">40703117</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Khunte</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sangha</surname><given-names>V</given-names> </name><name name-style="western"><surname>Oikonomou</surname><given-names>EK</given-names> </name><etal/></person-group><article-title>Detection of left ventricular systolic dysfunction from single-lead electrocardiography adapted for portable and wearable devices</article-title><source>NPJ Digit Med</source><year>2023</year><month>07</month><day>11</day><volume>6</volume><issue>1</issue><fpage>124</fpage><pub-id pub-id-type="doi">10.1038/s41746-023-00869-w</pub-id><pub-id pub-id-type="medline">37433874</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fleury</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Dubois</surname><given-names>R</given-names> </name><name name-style="western"><surname>Christophle-Boulard</surname><given-names>S</given-names> </name><name name-style="western"><surname>Extramiana</surname><given-names>F</given-names> </name><name name-style="western"><surname>Maison-Blanche</surname><given-names>P</given-names> </name></person-group><article-title>A deep learning modular ECG approach for cardiologist assisted adjudication of atrial fibrillation and atrial flutter episodes</article-title><source>Heart Rhythm O2</source><year>2024</year><month>12</month><volume>5</volume><issue>12</issue><fpage>862</fpage><lpage>872</lpage><pub-id pub-id-type="doi">10.1016/j.hroo.2024.09.007</pub-id><pub-id pub-id-type="medline">39803625</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Mitchell</surname><given-names>H</given-names> </name><name name-style="western"><surname>Rosario</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hernandez</surname><given-names>C</given-names> </name><name name-style="western"><surname>Lipsitz</surname><given-names>SR</given-names> </name><name name-style="western"><surname>Levine</surname><given-names>DM</given-names> </name></person-group><article-title>Single-lead arrhythmia detection through machine learning: cross-sectional evaluation of a novel algorithm using real-world data</article-title><source>Open Heart</source><year>2023</year><month>09</month><volume>10</volume><issue>2</issue><fpage>e002228</fpage><pub-id pub-id-type="doi">10.1136/openhrt-2022-002228</pub-id><pub-id pub-id-type="medline">37734747</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Clifford</surname><given-names>G</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Moody</surname><given-names>B</given-names> </name><etal/></person-group><article-title>AF classification from a short single lead ECG recording: the physionet computing in cardiology challenge 2017</article-title><conf-name>2017 Computing in Cardiology</conf-name><conf-date>Sep 24-27, 2017</conf-date><pub-id pub-id-type="doi">10.22489/CinC.2017.065-469</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Christov</surname><given-names>I</given-names> </name><name name-style="western"><surname>Krasteva</surname><given-names>V</given-names> </name><name name-style="western"><surname>Simova</surname><given-names>I</given-names> </name><name name-style="western"><surname>Neycheva</surname><given-names>T</given-names> </name><name name-style="western"><surname>Schmid</surname><given-names>R</given-names> </name></person-group><article-title>Ranking of the most reliable beat morphology and heart rate variability features for the detection of atrial fibrillation in short single-lead ECG</article-title><source>Physiol Meas</source><year>2018</year><month>09</month><day>24</day><volume>39</volume><issue>9</issue><fpage>094005</fpage><pub-id pub-id-type="doi">10.1088/1361-6579/aad9f0</pub-id><pub-id pub-id-type="medline">30102603</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Si</surname><given-names>J</given-names> </name><name name-style="western"><surname>Bao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>F</given-names> </name><etal/></person-group><article-title>Research on atrial fibrillation diagnosis in electrocardiograms based on CLA-AF model</article-title><source>Eur Heart J Digit Health</source><year>2025</year><month>01</month><volume>6</volume><issue>1</issue><fpage>82</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztae092</pub-id><pub-id pub-id-type="medline">39846071</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Ma</surname><given-names>H</given-names> </name><name name-style="western"><surname>Xia</surname><given-names>L</given-names> </name></person-group><article-title>Atrial fibrillation detection algorithm based on graph convolution network</article-title><source>IEEE Access</source><year>2023</year><volume>11</volume><fpage>67191</fpage><lpage>67200</lpage><pub-id pub-id-type="doi">10.1109/ACCESS.2023.3291352</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Feng</surname><given-names>K</given-names> </name><name name-style="western"><surname>Fan</surname><given-names>Z</given-names> </name></person-group><article-title>A novel bidirectional LSTM network based on scale factor for atrial fibrillation signals classification</article-title><source>Biomed Signal Process Control</source><year>2022</year><month>07</month><volume>76</volume><fpage>103663</fpage><pub-id pub-id-type="doi">10.1016/j.bspc.2022.103663</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salinas-Mart&#x00ED;nez</surname><given-names>R</given-names> </name><name name-style="western"><surname>de Bie</surname><given-names>J</given-names> </name><name name-style="western"><surname>Marzocchi</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sandberg</surname><given-names>F</given-names> </name></person-group><article-title>Detection of brief episodes of atrial fibrillation based on electrocardiomatrix and convolutional neural network</article-title><source>Front Physiol</source><year>2021</year><volume>12</volume><fpage>673819</fpage><pub-id pub-id-type="doi">10.3389/fphys.2021.673819</pub-id><pub-id pub-id-type="medline">34512372</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tarabanis</surname><given-names>C</given-names> </name><name name-style="western"><surname>Koesmahargyo</surname><given-names>V</given-names> </name><name name-style="western"><surname>Tachmatzidis</surname><given-names>D</given-names> </name><etal/></person-group><article-title>Artificial intelligence-enabled sinus electrocardiograms for the detection of paroxysmal atrial fibrillation benchmarked against the CHARGE-AF score</article-title><source>Eur Heart J Digit Health</source><year>2025</year><month>11</month><volume>6</volume><issue>6</issue><fpage>1134</fpage><lpage>1144</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztaf100</pub-id><pub-id pub-id-type="medline">41267852</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fang</surname><given-names>B</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>J</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>Dual-channel neural network for atrial fibrillation detection from a single lead ECG wave</article-title><source>IEEE J Biomed Health Inform</source><year>2023</year><month>05</month><volume>27</volume><issue>5</issue><fpage>2296</fpage><lpage>2305</lpage><pub-id pub-id-type="doi">10.1109/JBHI.2021.3120890</pub-id><pub-id pub-id-type="medline">34665746</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kisohara</surname><given-names>M</given-names> </name><name name-style="western"><surname>Masuda</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Yuda</surname><given-names>E</given-names> </name><name name-style="western"><surname>Ueda</surname><given-names>N</given-names> </name><name name-style="western"><surname>Hayano</surname><given-names>J</given-names> </name></person-group><article-title>Optimal length of R-R interval segment window for Lorenz plot detection of paroxysmal atrial fibrillation by machine learning</article-title><source>Biomed Eng Online</source><year>2020</year><month>06</month><day>16</day><volume>19</volume><issue>1</issue><fpage>49</fpage><pub-id pub-id-type="doi">10.1186/s12938-020-00795-y</pub-id><pub-id pub-id-type="medline">32546178</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yao</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Wu</surname><given-names>M</given-names> </name><etal/></person-group><article-title>Detection of atrial fibrillation using a nonlinear Lorenz Scattergram and deep learning in primary care</article-title><source>BMC Prim Care</source><year>2024</year><month>07</month><day>20</day><volume>25</volume><issue>1</issue><fpage>267</fpage><pub-id pub-id-type="doi">10.1186/s12875-024-02407-3</pub-id><pub-id pub-id-type="medline">39033295</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Weinberg</surname><given-names>KM</given-names> </name><name name-style="western"><surname>Denes</surname><given-names>P</given-names> </name><name name-style="western"><surname>Kadish</surname><given-names>AH</given-names> </name><name name-style="western"><surname>Goldberger</surname><given-names>JJ</given-names> </name></person-group><article-title>Development and validation of diagnostic criteria for atrial flutter on the surface electrocardiogram</article-title><source>Ann Noninvasive Electrocardiol</source><year>2008</year><month>04</month><volume>13</volume><issue>2</issue><fpage>145</fpage><lpage>154</lpage><pub-id pub-id-type="doi">10.1111/j.1542-474X.2008.00214.x</pub-id><pub-id pub-id-type="medline">18426440</pub-id></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>Hamilton</surname><given-names>P</given-names> </name></person-group><article-title>Open source ECG analysis</article-title><conf-name>Computers in Cardiology</conf-name><conf-date>Sep 22-25, 2002</conf-date><conf-loc>Memphis, TN, USA</conf-loc><fpage>101</fpage><lpage>104</lpage><pub-id pub-id-type="doi">10.1109/CIC.2002.1166717</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Uittenbogaart</surname><given-names>SB</given-names> </name><name name-style="western"><surname>Lucassen</surname><given-names>WAM</given-names> </name><name name-style="western"><surname>van Etten-Jamaludin</surname><given-names>FS</given-names> </name><name name-style="western"><surname>de Groot</surname><given-names>JR</given-names> </name><name name-style="western"><surname>van Weert</surname><given-names>H</given-names> </name></person-group><article-title>Burden of atrial high-rate episodes and risk of stroke: a systematic review</article-title><source>EP Europace</source><year>2018</year><month>09</month><day>1</day><volume>20</volume><issue>9</issue><fpage>1420</fpage><lpage>1427</lpage><pub-id pub-id-type="doi">10.1093/europace/eux356</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Healey</surname><given-names>JS</given-names> </name><name name-style="western"><surname>Connolly</surname><given-names>SJ</given-names> </name><name name-style="western"><surname>Gold</surname><given-names>MR</given-names> </name><etal/></person-group><article-title>Subclinical atrial fibrillation and the risk of stroke</article-title><source>N Engl J Med</source><year>2012</year><month>01</month><day>12</day><volume>366</volume><issue>2</issue><fpage>120</fpage><lpage>129</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa1105575</pub-id><pub-id pub-id-type="medline">22236222</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="confproc"><person-group person-group-type="author"><name name-style="western"><surname>He</surname><given-names>K</given-names> </name><name name-style="western"><surname>Zhang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Ren</surname><given-names>S</given-names> </name><name name-style="western"><surname>Sun</surname><given-names>J</given-names> </name></person-group><article-title>Deep residual learning for image recognition</article-title><conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name><conf-date>Jun 27-30, 2016</conf-date><conf-loc>Las Vegas, NV, USA</conf-loc><fpage>770</fpage><lpage>778</lpage><pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Seo</surname><given-names>HC</given-names> </name><name name-style="western"><surname>Oh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>H</given-names> </name><name name-style="western"><surname>Joo</surname><given-names>S</given-names> </name></person-group><article-title>ECG data dependency for atrial fibrillation detection based on residual networks</article-title><source>Sci Rep</source><year>2021</year><month>09</month><day>14</day><volume>11</volume><issue>1</issue><fpage>18256</fpage><pub-id pub-id-type="doi">10.1038/s41598-021-97308-1</pub-id><pub-id pub-id-type="medline">34521892</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Patel</surname><given-names>S</given-names> </name><name name-style="western"><surname>Wang</surname><given-names>M</given-names> </name><name name-style="western"><surname>Guo</surname><given-names>J</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>G</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>C</given-names> </name></person-group><article-title>A study of R-R interval transition matrix features for machine learning algorithms in AFib detection</article-title><source>Sensors</source><year>2023</year><month>04</month><day>3</day><volume>23</volume><issue>7</issue><fpage>3700</fpage><pub-id pub-id-type="doi">10.3390/s23073700</pub-id><pub-id pub-id-type="medline">37050761</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lown</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>M</given-names> </name><name name-style="western"><surname>Brown</surname><given-names>C</given-names> </name><etal/></person-group><article-title>Machine learning detection of atrial fibrillation using wearable technology</article-title><source>PLOS ONE</source><year>2020</year><volume>15</volume><issue>1</issue><fpage>e0227401</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0227401</pub-id><pub-id pub-id-type="medline">31978173</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Krasteva</surname><given-names>V</given-names> </name><name name-style="western"><surname>Stoyanov</surname><given-names>T</given-names> </name><name name-style="western"><surname>Naydenov</surname><given-names>S</given-names> </name><name name-style="western"><surname>Schmid</surname><given-names>R</given-names> </name><name name-style="western"><surname>Jekova</surname><given-names>I</given-names> </name></person-group><article-title>Detection of atrial fibrillation in holter ECG recordings by ECHOView images: a deep transfer learning study</article-title><source>Diagnostics</source><year>2025</year><month>03</month><day>28</day><volume>15</volume><issue>7</issue><fpage>865</fpage><pub-id pub-id-type="doi">10.3390/diagnostics15070865</pub-id><pub-id pub-id-type="medline">40218215</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Rawshani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Rawshani</surname><given-names>A</given-names> </name><name name-style="western"><surname>Smith</surname><given-names>G</given-names> </name><etal/></person-group><article-title>Integrating deep learning with ECG, heart rate variability and demographic data for improved detection of atrial fibrillation</article-title><source>Open Heart</source><year>2025</year><month>03</month><day>31</day><volume>12</volume><issue>1</issue><fpage>e003185</fpage><pub-id pub-id-type="doi">10.1136/openhrt-2025-003185</pub-id><pub-id pub-id-type="medline">40164487</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xintarakou</surname><given-names>A</given-names> </name><name name-style="western"><surname>Sousonis</surname><given-names>V</given-names> </name><name name-style="western"><surname>Asvestas</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vardas</surname><given-names>PE</given-names> </name><name name-style="western"><surname>Tzeis</surname><given-names>S</given-names> </name></person-group><article-title>Remote cardiac rhythm monitoring in the era of smart wearables: present assets and future perspectives</article-title><source>Front Cardiovasc Med</source><year>2022</year><volume>9</volume><fpage>853614</fpage><pub-id pub-id-type="doi">10.3389/fcvm.2022.853614</pub-id><pub-id pub-id-type="medline">35299975</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Burykin</surname><given-names>A</given-names> </name><name name-style="western"><surname>Costa</surname><given-names>MD</given-names> </name><name name-style="western"><surname>Citi</surname><given-names>L</given-names> </name><name name-style="western"><surname>Goldberger</surname><given-names>AL</given-names> </name></person-group><article-title>Dynamical density delay maps: simple, new method for visualising the behaviour of complex systems</article-title><source>BMC Med Inform Decis Mak</source><year>2014</year><month>01</month><day>18</day><volume>14</volume><issue>1</issue><fpage>6</fpage><pub-id pub-id-type="doi">10.1186/1472-6947-14-6</pub-id><pub-id pub-id-type="medline">24438439</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Zhao</surname><given-names>L</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>C</given-names> </name><name name-style="western"><surname>Wei</surname><given-names>S</given-names> </name><name name-style="western"><surname>Shen</surname><given-names>Q</given-names> </name><name name-style="western"><surname>Zhou</surname><given-names>F</given-names> </name><name name-style="western"><surname>Li</surname><given-names>J</given-names> </name></person-group><article-title>A new entropy-based atrial fibrillation detection method for scanning wearable ECG recordings</article-title><source>Entropy</source><year>2018</year><month>11</month><day>26</day><volume>20</volume><issue>12</issue><fpage>904</fpage><pub-id pub-id-type="doi">10.3390/e20120904</pub-id><pub-id pub-id-type="medline">33266628</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jin</surname><given-names>X</given-names> </name><name name-style="western"><surname>Hirakawa</surname><given-names>K</given-names> </name></person-group><article-title>Analysis and processing of pixel binning for color image sensor</article-title><source>EURASIP J Adv Signal Process</source><year>2012</year><month>12</month><volume>2012</volume><issue>1</issue><fpage>125</fpage><pub-id pub-id-type="doi">10.1186/1687-6180-2012-125</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>JY</given-names> </name><name name-style="western"><surname>Kim</surname><given-names>KG</given-names> </name><name name-style="western"><surname>Tae</surname><given-names>Y</given-names> </name><etal/></person-group><article-title>An artificial intelligence algorithm with 24-h Holter monitoring for the identification of occult atrial fibrillation during sinus rhythm</article-title><source>Front Cardiovasc Med</source><year>2022</year><volume>9</volume><fpage>906780</fpage><pub-id pub-id-type="doi">10.3389/fcvm.2022.906780</pub-id><pub-id pub-id-type="medline">35872911</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hennings</surname><given-names>E</given-names> </name><name name-style="western"><surname>Coslovsky</surname><given-names>M</given-names> </name><name name-style="western"><surname>Paladini</surname><given-names>RE</given-names> </name><etal/></person-group><article-title>Assessment of the atrial fibrillation burden in Holter electrocardiogram recordings using artificial intelligence</article-title><source>Cardiovasc Digit Health J</source><year>2023</year><month>04</month><volume>4</volume><issue>2</issue><fpage>41</fpage><lpage>47</lpage><pub-id pub-id-type="doi">10.1016/j.cvdhj.2023.01.003</pub-id><pub-id pub-id-type="medline">37101946</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Domazetoski</surname><given-names>V</given-names> </name><name name-style="western"><surname>Gligoric</surname><given-names>G</given-names> </name><name name-style="western"><surname>Marinkovic</surname><given-names>M</given-names> </name><etal/></person-group><article-title>The influence of atrial flutter in automated detection of atrial arrhythmias - are we ready to go into clinical practice?"</article-title><source>Comput Methods Programs Biomed</source><year>2022</year><month>06</month><volume>221</volume><fpage>106901</fpage><pub-id pub-id-type="doi">10.1016/j.cmpb.2022.106901</pub-id><pub-id pub-id-type="medline">35636359</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Fiorina</surname><given-names>L</given-names> </name><name name-style="western"><surname>Chemaly</surname><given-names>P</given-names> </name><name name-style="western"><surname>Cellier</surname><given-names>J</given-names> </name><etal/></person-group><article-title>Artificial intelligence-based electrocardiogram analysis improves atrial arrhythmia detection from a smartwatch electrocardiogram</article-title><source>Eur Heart J Digit Health</source><year>2024</year><month>09</month><volume>5</volume><issue>5</issue><fpage>535</fpage><lpage>541</lpage><pub-id pub-id-type="doi">10.1093/ehjdh/ztae047</pub-id><pub-id pub-id-type="medline">39318690</pub-id></nlm-citation></ref><ref id="ref44"><label>44</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>De Guio</surname><given-names>F</given-names> </name><name name-style="western"><surname>Rienstra</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lillo-Castellano</surname><given-names>JM</given-names> </name><etal/></person-group><article-title>Enhanced detection of atrial fibrillation in single-lead electrocardiograms using a Cloud-based artificial intelligence platform</article-title><source>Heart Rhythm</source><year>2025</year><month>07</month><volume>22</volume><issue>7</issue><fpage>1667</fpage><lpage>1674</lpage><pub-id pub-id-type="doi">10.1016/j.hrthm.2024.12.048</pub-id><pub-id pub-id-type="medline">39800092</pub-id></nlm-citation></ref><ref id="ref45"><label>45</label><nlm-citation citation-type="web"><source>GitHub</source><access-date>2026-06-30</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://github.com/YasuokaMasanao/ECG_AnnotationTool">https://github.com/YasuokaMasanao/ECG_AnnotationTool</ext-link></comment></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Supplementary figures illustrating RRI histogram representations, representative noise-labeled ECG segments, one-vs-rest ROC curves, and external-validation examples for the 3-class AF screening model.</p><media xlink:href="medinform_v14i1e91960_app1.pdf" xlink:title="PDF File, 3916 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Supplementary tables summarizing patient-level AF and noise burden, ectopic burden assessment, annotation agreement, detailed model performance metrics, subgroup analysis by garment type, and secondary external validation results.</p><media xlink:href="medinform_v14i1e91960_app2.docx" xlink:title="DOCX File, 28 KB"/></supplementary-material></app-group></back></article>