﻿<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.0 20120330//EN" "http://jats.nlm.nih.gov/publishing/1.0/JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="nlm-ta">J Cardiovasc Aging.</journal-id>
      <journal-id journal-id-type="publisher-id">JCA</journal-id>
      <journal-title-group>
        <journal-title>The Journal of Cardiovascular Aging</journal-title>
      </journal-title-group>
      <issn pub-type="epub">2768-5993</issn>
      <publisher>
        <publisher-name>OAE Publishing Inc.</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
	<article-id pub-id-type="doi">10.20517/jca.2026.46</article-id>
      <article-categories>
        <subj-group>
          <subject>Original Research Article</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Morphological analysis and multimodal deep learning for impedance cardiography feature localization and hemodynamic parameter estimation</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <name>
            <surname>Ma</surname>
            <given-names>Shuai</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
          <xref ref-type="aff" rid="I#">
            <sup>#</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Xu</surname>
            <given-names>Zeyue</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
          <xref ref-type="aff" rid="I#">
            <sup>#</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Yan</surname>
            <given-names>Zhengxu</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Wu</surname>
            <given-names>Ningxia</given-names>
          </name>
          <xref ref-type="aff" rid="I3">
            <sup>3</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Zhao</surname>
            <given-names>Zhe</given-names>
          </name>
          <xref ref-type="aff" rid="I1">
            <sup>1</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Wang</surname>
            <given-names>Huiquan</given-names>
          </name>
          <xref ref-type="aff" rid="I2">
            <sup>2</sup>
          </xref>
          <xref ref-type="aff" rid="I4">
            <sup>4</sup>
          </xref>
		  <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-7896-303X</contrib-id>
          <xref ref-type="corresp" rid="cor1" />
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <name>
            <surname>Cao</surname>
            <given-names>Feng</given-names>
          </name>
          <xref ref-type="aff" rid="I5">
            <sup>5</sup>
          </xref>
          <xref ref-type="aff" rid="I6">
            <sup>6</sup>
          </xref>
		  <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-1010-6429</contrib-id>
          <xref ref-type="corresp" rid="cor1" />
        </contrib>
      </contrib-group>
      <aff id="I1">
        <sup>1</sup>School of Electronic and Information Engineering, Tiangong University, Tianjin 300387, China.</aff>
      <aff id="I2">
        <sup>2</sup>School of Life Sciences, Tiangong University, Tianjin 300387, China.</aff>
      <aff id="I3">
        <sup>3</sup>Department of Medicine, Nankai University, Tianjin 300071, China.</aff>
      <aff id="I4">
        <sup>4</sup>Tianjin Key Laboratory of Quality Control and Evaluation Technology for Medical Devices, Tianjin 300384, China.</aff>
      <aff id="I5">
        <sup>5</sup>Institute of Geriatric Medicine, Beijing Key Laboratory of Geriatric Comorbidity, National Clinical Research Center for Geriatric Diseases, Chinese PLA General Hospital, Beijing 100039, China.</aff>
      <aff id="I6">
        <sup>6</sup>National Key Laboratory of Kidney Diseases, National Clinical Research Center for Chronic Kidney Diseases, Chinese PLA General Hospital, Beijing 100039, China.</aff>
      <aff id="I#"><sup>#</sup>Authors contributed equally.</aff>
      <author-notes>
        <corresp id="cor1">Correspondence to: Prof./Dr. Feng Cao, Institute of Geriatric Medicine, Beijing Key Laboratory of Geriatric Comorbidity, National Clinical Research Center for Geriatric Diseases, Chinese PLA General Hospital, Beijing 100039, China. E-mail: <email>caofeng@301hospital.com.cn</email>; Prof./Dr. Huiquan Wang,  School of Life Sciences, Tiangong University, Tianjin 300387, China. E-mail: <email>huiquan@tiangong.edu.cn</email></corresp>
     
	   <fn fn-type="other">
          <p>
            <bold>Received:</bold> 27 Apr 2025 | <bold>First Decision:</bold> 18 Jun 2026 | <bold>Revised:</bold> 30 Jun 2026 | <bold>Accepted:</bold> 28 Jul 2026 | <bold>Published:</bold> 18 Aug 2026</p>
        </fn>
        <fn fn-type="other">
          <p>
            <bold>Academic Editor:</bold> Houzao Chen | <bold>Copy Editor:</bold> Fangling Lan |  <bold>Production Editor:</bold> Fangling Lan</p>
        </fn>
      </author-notes>
	  <pub-date pub-type="ppub">
        <year>2026</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>18</day>
        <month>8</month>
        <year>2026</year>
      </pub-date>
      <volume>6</volume>
	  <issue>3</issue>
      <elocation-id>30</elocation-id>
	 
	 
	 
      <permissions>
        <copyright-statement>© The Author(s) 2026.</copyright-statement>
        <license xlink:href="https://creativecommons.org/licenses/by/4.0/">
          <license-p>© The Author(s) 2026. <bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License (<uri xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</uri>), which permits unrestricted use, sharing, adaptation, distribution and reproduction in any medium or format, for any purpose, even commercially, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons license, and indicate if changes were made.</license-p>
        </license>
      </permissions>
      <abstract>
        <p>
          <bold>Aim:</bold> This study aimed to reduce the susceptibility of impedance cardiography (ICG) feature-point localization to waveform variability, noise, and inter-individual differences by developing an Electrocardiography (ECG)-ICG-phonocardiography (PCG) multimodal framework for beat-to-beat estimation of left ventricular ejection time (LVET), stroke volume (SV), and cardiac output (CO).</p>
        <p>
          <bold>Methods:</bold> Using the public HeartCycle multimodal dataset, ICG waveform polymorphism and class overlap were characterized using dynamic time warping, K-medoids clustering, and morphological classification. We then developed an end-to-end model with modality-specific one-dimensional convolutional encoders, Transformer-based temporal modeling, and event-query localization. ECG provided beat alignment, and the PCG envelope supplied mechanical-event timing information. Performance was compared with conventional rule-based methods, learning-based baselines, and ablation models under a unified evaluation framework, with paired analyses and leave-one-subject-out (LOSO) validation.</p>
        <p>
          <bold>Results:</bold> Increasing clustering granularity increased morphological overlap and classification difficulty, indicating limitations of local ICG-only rules. On the primary test set, the PCG-integrated model achieved a mean absolute percentage error of 6.51% for SV and 7.35% for CO, lower than those of the conventional rule-based method and the deep-learning model without PCG. Subject-level paired analyses based on the 17-fold LOSO results demonstrated significant reductions in SV and CO estimation errors following PCG integration and further supported the feasibility of subject-independent beat-level estimation. The complete model also yielded lower errors than learning-based baselines and ablation models under current evaluation conditions. These findings reflect the performance of the integrated framework, while its generalizability warrants further validation.</p>
        <p>
          <bold>Conclusion:</bold> ECG-ICG-PCG fusion provides a feasible strategy for more stable beat-to-beat LVET, SV, and CO estimation under the current evaluation conditions. Larger multicenter datasets, diverse patient populations, and prospective longitudinal studies are needed to establish reliable trend monitoring and within-subject tracking of hemodynamic changes.</p>
      </abstract>
      <kwd-group>
        <kwd>Impedance cardiography</kwd>
        <kwd>phonocardiography</kwd>
        <kwd>multimodal deep learning</kwd>
        <kwd>feature-point localization</kwd>
        <kwd>hemodynamic parameter estimation</kwd>
        <kwd>waveform morphology analysis</kwd>
      </kwd-group>
    </article-meta>
  </front>
  <body>
    <sec id="sec1">
      <title>INTRODUCTION</title>
      <p>In clinical settings such as critical care, perioperative management, and longitudinal follow-up of chronic cardiovascular disease, continuous hemodynamic monitoring is essential for assessing volume status, guiding circulatory support, and evaluating therapeutic efficacy. Compared with invasive approaches, including pulmonary artery catheterization, impedance cardiography (ICG) enables noninvasive, continuous, beat-to-beat hemodynamic assessment by tracking dynamic changes in thoracic electrical impedance throughout the cardiac cycle. Consequently, it has shown considerable promise for monitoring key parameters such as stroke volume (SV), cardiac output (CO), and left ventricular ejection time (LVET). Since the seminal work of Kubicek and colleagues, who introduced a method for estimating CO based on thoracic impedance variations, ICG has been extensively explored in both clinical research and engineering applications, and is increasingly regarded as a promising candidate for wearable, continuous cardiac function monitoring<sup>[<xref ref-type="bibr" rid="B1">1</xref>-<xref ref-type="bibr" rid="B6">6</xref>]</sup>. However, the physiological origins of the ICG signal are inherently complex. Its waveform reflects not only aortic blood flow dynamics but also the combined influence of large-vessel volume changes, blood acceleration, and variations in the conductive properties of thoracic tissues<sup>[<xref ref-type="bibr" rid="B7">7</xref>]</sup>. As a result, characteristic points in the ICG waveform do not consistently exhibit a stable one-to-one correspondence with specific mechanical cardiac events. This intrinsic variability renders the computation of ICG-derived hemodynamic parameters highly dependent on the accuracy and robustness of feature-point localization.</p>
      <p>Existing studies on key feature points in the ICG waveform, particularly the B and X points associated with aortic valve opening (AVO) and closure, have largely relied on threshold-crossing rules, local extremum detection, template matching, or empirically defined constraints for event localization, from which LVET, SV, and CO are subsequently derived<sup>[<xref ref-type="bibr" rid="B8">8</xref>-<xref ref-type="bibr" rid="B12">12</xref>]</sup>. Although such approaches retain a degree of interpretability under ideal signal conditions, they are fundamentally predicated on the assumption that local waveform morphology maps reliably onto the underlying physiological events. In real-world settings, however, respiration, motion artifacts, postural changes, inter-individual variability, and measurement noise can all substantially distort the local structure of the ICG waveform, rendering the definitions of the B and X points increasingly ambiguous and thereby introducing temporal drift or even cross-beat misidentification. In recent years, a number of studies have sought to improve the stability of feature-point extraction by enhancing ICG denoising and signal decomposition through methods such as coherence analysis, canonical correlation analysis, ICEEMDAN, variational mode decomposition (VMD), and related refinements<sup>[<xref ref-type="bibr" rid="B13">13</xref>-<xref ref-type="bibr" rid="B17">17</xref>]</sup>. While these techniques can improve signal quality to some extent, most still follow a serial processing paradigm of denoising first, point detection second, and parameter estimation last, such that errors may accumulate progressively across intermediate steps. Moreover, in beat-to-beat estimation and cross-subject generalization, dependence on single-modality ICG waveforms and local morphological heuristics remains insufficient to resolve feature-point ambiguity at its source<sup>[<xref ref-type="bibr" rid="B18">18</xref>]</sup>. Therefore, in ICG analysis, it is not enough to assess whether detected points conform to morphological definitions alone; a more clinically meaningful perspective is to determine whether localization errors propagate downstream and ultimately compromise hemodynamic parameter estimation.</p>
      <p>From a physiological perspective, a single-modality ICG signal provides only a limited representation of mechanical ejection events, whereas multimodal physiological signal integration offers a more comprehensive solution. Electrocardiography (ECG) provides a stable reference for the onset of cardiac electrical activity and can serve as a reliable anchor for beat segmentation and inter-cycle alignment. In contrast, phonocardiography (PCG) directly captures mechanical vibrations generated by valve closure and blood flow, with the first and second heart sounds exhibiting well-defined temporal relationships with key systolic mechanical events. As such, PCG has the potential to impose additional constraints on events associated with AVO and closure<sup>[<xref ref-type="bibr" rid="B19">19</xref>-<xref ref-type="bibr" rid="B25">25</xref>]</sup>. Compared with single-modality ICG analysis, joint modeling of ECG, ICG, and PCG enables the establishment of a more complete temporal correspondence across electrical activation, mechanical vibration, and impedance variation. This integrated representation can mitigate the uncertainties introduced by noise, morphological distortion, and inter-individual variability, thereby improving the robustness of feature interpretation and downstream hemodynamic assessment.</p>
      <p>Building on the above considerations, this study establishes an integrated research framework that encompasses ICG waveform morphology analysis and multimodal deep fusion modeling. First, using the HeartCycle multimodal dataset, we conducted a quantitative investigation of ICG waveforms through unsupervised clustering and morphological classification, thereby revealing pronounced morphological heterogeneity and substantial overlap between waveform categories. These findings provide data-driven evidence for the intrinsic limitations of single-modality morphological rules. Second, we developed an end-to-end deep learning model that integrates ECG, ICG, and PCG. In this framework, the PCG envelope is used to provide temporal priors for mechanical cardiac events, while one-dimensional convolutional encoding, Transformer-based global temporal modeling, and an event-query mechanism are jointly leveraged to achieve robust localization of the onset and termination of cardiac ejection. Finally, by introducing an auxiliary regression branch to impose physiological consistency constraints, the localized feature points are further mapped to beat-to-beat estimates of LVET, SV, and CO, enabling a unified comparison between conventional rule-based approaches and multimodal deep learning methods across the entire analytical pipeline. Overall, this work seeks to advance ICG analysis from local morphology-based feature detection toward multimodal modeling optimized for functional parameter estimation, and to provide a methodological foundation for the development of continuous, noninvasive cardiac function monitoring.</p>
    </sec>
    <sec id="sec2">
      <title>MATERIALS AND METHODS</title>
      <p>This study established a complete analytical pipeline for ICG feature-point localization and hemodynamic parameter estimation, comprising data preprocessing, waveform morphology analysis, feature-point detection, and parameter derivation. First, using the HeartCycle multimodal dataset, ECG, ICG, and PCG signals were synchronously preprocessed, segmented on a beat-by-beat basis, and normalized to construct a unified beat-level multimodal input. Next, unsupervised clustering and morphological classification were performed on the ICG waveforms to quantitatively characterize their morphological heterogeneity and category overlap. On this basis, the performance of conventional rule-based approaches and deep learning methods was further compared in both key feature-point localization and beat-to-beat hemodynamic parameter estimation. To highlight the contribution of multimodal information - particularly the temporal priors for mechanical events provided by PCG - the study design incorporated multiple comparative settings, including single-modality versus multimodality and rule-based versus learning-based approaches.</p>
      <sec id="sec2-1">
        <title>Data source and signal preprocessing</title>
        <p>This study was conducted using the HeartCycle v1.0.0 multimodal dataset, which is publicly available through PhysioNet (<uri xlink:href="https://physionet.org/content/heartcycle/1.0.0/">https://physionet.org/content/heartcycle/1.0.0/</uri>)<sup>[<xref ref-type="bibr" rid="B26">26</xref>]</sup>. The dataset comprises synchronously acquired signals from 17 healthy subjects, including ECG, ICG, PCG, and ultrasound-derived reference information, as illustrated in <xref ref-type="fig" rid="fig1">Figure 1</xref>. These data provide a reliable basis for cardiac cycle-level event localization and hemodynamic parameter evaluation. Unlike studies relying solely on ICG, the HeartCycle dataset offers concurrent observations of electrical activity, mechanical vibration, and impedance variation, making it particularly well suited for multimodal temporal alignment and fusion modeling. All model training, validation, and testing in this study were performed exclusively on this dataset, without the inclusion of external data sources.</p>
        <fig id="fig1" position="float">
          <label>Figure 1</label>
          <caption>
            <p>Example of synchronized raw multimodal segments in the HeartCycle dataset: (A) ECG; (B) ICG; (C) PCG; (D) echocardiography.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.1.jpg" />
        </fig>
        <p>To preserve physiologically relevant components while suppressing noise, modality-specific preprocessing was performed. For ECG, the conventional rule-based analysis first used a 50 Hz notch filter to suppress power-line interference (Q = 30), followed by a fourth-order Butterworth 0.5-40 Hz band-pass filter to remove baseline wander and high-frequency electromyographic noise. In the deep learning model, ECG was resampled and standardized beat by beat as the temporal reference for electrical activation. For ICG, the conventional morphological analysis first applied a fourth-order Butterworth 0.7-20 Hz band-pass filter to the impedance signal Z, and then used a Savitzky-Golay filter to compute the first derivative dZ/dt<sup>[<xref ref-type="bibr" rid="B27">27</xref>]</sup>. The S-G differentiation window length was 11 samples (approximately 55 ms at 200 Hz), the polynomial order was 3, and the derivative step was delta = 1/200 s. A fourth-order Butterworth 25 Hz low-pass filter was subsequently applied to ICG to suppress residual high-frequency noise. In the deep learning model, the ICG channel was represented by the first gradient of the resampled impedance signal and standardized by z-score normalization within each beat window. For PCG, the conventional ECG + PCG-assisted pipeline used a fourth-order Butterworth 20-200 Hz band-pass filter and smoothed the downsampled amplitude envelope with a 15 Hz low-pass filter. In the deep learning model, PCG was resampled to 4,000 Hz when necessary, filtered using a fourth-order Butterworth 20-800 Hz band-pass filter, transformed into an analytic-signal amplitude envelope by the Hilbert transform<sup>[<xref ref-type="bibr" rid="B28">28</xref>,<xref ref-type="bibr" rid="B29">29</xref>]</sup>, and resampled to 200 Hz for strict synchronization with ECG and ICG. After single-modality preprocessing, beat-level alignment and segmentation were performed using the ECG R peak as the temporal anchor. The main deep learning model used a fixed input window from 0.20 s before to 0.80 s after the R peak, with a unified sampling rate of 200 Hz, corresponding to 200 time points per beat, as shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. To further reduce the impact of amplitude differences among subjects on subsequent morphological analysis and model training, all segmented heartbeat segments underwent amplitude normalization or z-score standardization. After these steps, a multimodal beat-level input with strict temporal alignment, controlled noise levels, and clear physiological significance was obtained, laying a unified data foundation for subsequent ICG waveform morphological analysis and event localization based on multimodal constraints.</p>
        <fig id="fig2" position="float">
          <label>Figure 2</label>
          <caption>
            <p>Example of ICG feature-point localization under multimodal denoising constraints: (A) ICG; (B) ECG; (C) PCG.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.2.jpg" />
        </fig>
        <p>The reference time points for AVO and aortic valve closure (AVC) were obtained directly from the event annotations publicly released with the HeartCycle dataset rather than manually relabeled in this study. Specifically, the AVO and AVC event coordinates provided in the HeartCycle measurement files were used as the common reference standard for the onset and termination of ventricular ejection. The algorithmic B and X points reported in this study therefore denote feature locations predicted on the ICG waveform by rule-based or learning-based methods, whereas their supervised training and evaluation were based on the dataset-provided AVO and AVC reference labels.</p>
        <p>Because the present work relies on public dataset annotations, potential label uncertainty may arise from ultrasound-derived event identification, multimodal synchronization, finite sampling resolution, beat matching, and small timing differences between waveform-derived ICG landmarks and true valvular mechanical events. Such uncertainty can affect feature-point localization and propagate into LVET estimation through the B-X interval. It may also influence downstream SV and CO estimates because these parameters are derived from LVET, heart rate, and ICG morphological descriptors. For this reason, the present study emphasizes parameter-level evaluation in addition to point-localization performance, and reports error distributions, confidence intervals, and paired statistical tests to characterize the impact of label and localization uncertainty on final hemodynamic estimates.</p>
      </sec>
      <sec id="sec2-2">
        <title>Unsupervised clustering and morphological classification of ICG waveforms</title>
        <p>To systematically characterize morphological variation in ICG waveforms across cardiac cycles, unsupervised clustering and morphological classification analyses were performed on the valid preprocessed ICG beat segments. Given that physiological signals commonly exhibit local temporal stretching, phase shifts, and timing misalignment, dynamic time warping (DTW) distance was adopted to quantify waveform similarity and to construct the pairwise distance matrix. Compared with conventional point-to-point Euclidean distance, DTW is better suited for evaluating the overall similarity of ICG waveforms in the presence of temporal fluctuation<sup>[<xref ref-type="bibr" rid="B30">30</xref>]</sup>.</p>
        <p>Based on the resulting distance matrix, K-medoids clustering was applied to the ICG segments to identify latent waveform morphological subtypes<sup>[<xref ref-type="bibr" rid="B31">31</xref>]</sup>. To examine the distribution of waveform patterns under different levels of clustering granularity, the number of clusters, K, was set to 2, 3, 4, and 5, and the class composition and waveform differences under each clustering scheme were compared. It should be emphasized that the purpose of this analysis was not to establish a fixed clinical taxonomy, but rather to determine, from a data-driven perspective, whether single-modality ICG waveforms exhibit substantial morphological heterogeneity and feature overlap. This analysis also provides a basis for understanding the limitations of conventional rule-based methods in subsequent investigations.</p>
        <p>To support both the discovery of ICG waveform morphologies and the evaluation of classification performance, this study conducts unsupervised morphological clustering and supervised classification at two separate levels. First, DTW-K-Medoids clustering is applied to all 1,908 valid ICG beat segments from 17 subjects to characterize the overall morphological heterogeneity of the ICG waveforms in the HeartCycle cohort, identify representative waveforms, and examine the class composition under different numbers of clusters. This step describes the morphological patterns present in the study cohort and defines candidate morphology classes for the subsequent classification tasks. On this basis, the supervised classification experiments adopt a fixed subject-level split to evaluate model generalization to unseen subjects. No subject overlap exists among the training, validation, and test sets, ensuring that multiple beat segments from the same subject do not appear in different data subsets. For each <italic>K</italic> value, the clustering prototypes, data standardization parameters, and class weights are determined exclusively from the training set. The validation and test sets are transformed, assigned labels, and evaluated using only the parameters obtained during training.</p>
        <p>The same fixed subject-level data split is used for all classification tasks with K = 2, K = 3, K = 4, and K = 5 to ensure comparability across different numbers of clusters and model architectures. The training set contains 1,145 beat segments from 11 subjects, the validation set contains 381 beat segments from 3 subjects, and the test set contains 382 beat segments from 3 subjects. The three subsets are completely disjoint at the subject level, and all valid beat segments from each subject are assigned to only one subset. The fixed split comprising 11 training subjects, 3 validation subjects, and 3 test subjects is used exclusively for the ICG morphology classification experiments and is not applied to the subsequent LVET, SV, and CO estimation experiments. Morphology classification and hemodynamic parameter estimation use independent data construction and model training pipelines. Therefore, the subject composition and number of beat segments should be interpreted separately for the two tasks. To prevent information leakage from the test set into model training, all input standardization parameters are estimated using only the training set. Specifically, the mean <italic>μ<sub>train</sub></italic>(<italic>t</italic>) and standard deviation <italic>σ<sub>train</sub></italic>(<italic>t</italic>) are calculated at each time point t from the training data, and the same training-derived statistics are then used to standardize the training, validation, and test sets. For each <italic>K</italic> value, the K-Medoids clustering prototypes are fitted using only the training beat segments. Morphology labels for the validation and test segments are determined according to their DTW distances to the training-derived prototypes, with each segment assigned to the nearest prototype. Class weights are also calculated exclusively from the class frequencies in the training set. Let <italic>n<sub>c</sub></italic> denote the number of training samples in class <italic>c,</italic> and <italic>N<sub>train</sub></italic> denote the total number of training samples. The initial weight for class c is defined as <italic>N<sub>train</sub></italic>/<italic>n<sub>c</sub></italic>, after which all class weights are divided by their mean so that the normalized weights have an average value of 1. These weights are used only in the weighted cross-entropy loss during training, while the class frequencies in the validation and test sets are not involved in their calculation. The validation set is used only for model selection and for saving the model parameters associated with the best validation performance. The test set is evaluated only once after model training and parameter selection are complete.</p>
        <p>To further justify the selection of the number of clusters, the clustering results obtained with K = 2, 3, 4, and 5 were evaluated in terms of validity and stability. Specifically, for each value of K, K-Medoids clustering was repeated using five different random seeds, and several clustering validity metrics were calculated, including the Silhouette score, Calinski-Harabasz index, Davies-Bouldin index, inter-/intra-cluster distance ratio, and minimum cluster proportion. To assess the sensitivity of the clustering results to random initialization, the adjusted Rand index (ARI) and normalized mutual information (NMI) were further used to evaluate label consistency across different random seeds. For each value of K, the five repeated clustering runs generated a total of 10 pairwise comparisons, which were used to quantify the consistency and stability of the cluster assignments under different random seeds.</p>
        <p>Based on the clustering labels, we further developed an ICG morphology classification model, MS-GBiLSTM-TA, to evaluate the learnability of waveform features under different clustering granularities, as illustrated in <xref ref-type="fig" rid="fig3">Figure 3</xref>. The model takes a single ICG segment as input and employs a multi-branch convolutional architecture to extract local morphological features at different scales. A bidirectional long short-term memory network is then used to capture contextual dependencies along the temporal dimension. Subsequently, a temporal attention mechanism is introduced to adaptively weight informative time intervals, thereby enhancing discriminative capability for complex waveform patterns. Because the sample distribution was imbalanced across different clustering schemes, a weighted training strategy was adopted to mitigate the influence of class imbalance on classification performance. By comparing classification results across different clustering granularities, the inter-class separability of ICG waveforms and the clarity of their morphological boundaries could be further assessed.</p>
        <fig id="fig3" position="float">
          <label>Figure 3</label>
          <caption>
            <p>Architecture of the MS-GBiLSTM-TA classification model.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.3.jpg" />
        </fig>
      </sec>
      <sec id="sec2-3">
        <title>Multimodal feature localization and parameter estimation framework with PCG constraints</title>
        <p>This study adopts a single cardiac cycle (beat) as the fundamental unit of analysis. This design is motivated by the intrinsic periodicity of key ICG waveform components, including the systolic upstroke, incisura, and diastolic decay. In both clinical and engineering practice, hemodynamic parameters - such as LVET, SV, and CO - are typically reported on a beat-to-beat basis or through sliding-window averaging. Accordingly, defining model inputs, performing feature-point localization, and deriving parameters at the beat level not only aligns with real-world application requirements but also mitigates the influence of inter-cycle drift on model performance and estimation stability.</p>
        <p>For the <italic>b</italic>-th cardiac cycle, the synchronized multimodal input segment is defined as</p>
        <p><disp-formula> <label>(1)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} \mathbf{X}^{(b)}=\left[\mathbf{x}_{\mathrm{ICG}}^{(b)}, \mathbf{x}_{\mathrm{ECG}}^{(b)}, \mathbf{x}_{\mathrm{PCG}}^{(b)}\right], \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>where <inline-formula><tex-math id="M4">$$\mathbf{x}_{\mathrm{ICG}}^{(b)}$$</tex-math></inline-formula>, <inline-formula><tex-math id="M4">$$\mathbf{x}_{\mathrm{ECG}}^{(b)}$$</tex-math></inline-formula> and <inline-formula><tex-math id="M4">$$\mathbf{x}_{\mathrm{PCG}}^{(b)}$$</tex-math></inline-formula> denote the time-domain segments of the ICG, ECG, and PCG signals, respectively, within the <italic>b</italic>-th cardiac cycle. To ensure comparability across subjects and recordings, all modalities were cropped into fixed-length beat windows, with the ECG R peak serving as the temporal anchor for alignment. This alignment strategy preserves a consistent phase reference across beats, thereby reducing the influence of temporal offsets on model learning and preventing alignment errors from being misinterpreted as physiological variation. Within each input window, the first derivative of the ICG signal was used as the primary representation to emphasize dynamic morphological changes associated with mechanical ejection events; the ECG signal provided temporal reference for cardiac electrical activity, whereas the PCG signal supplied complementary acoustic information on valvular mechanical events.</p>
        <p>This study compares four feature-point localization strategies, each formulated in a unified manner as a mapping from multimodal inputs to the temporal positions of key feature points. For the <italic>b</italic>-th cardiac cycle, the <italic>m</italic>-th localization strategy can be expressed as</p>
       <p><disp-formula> <label>(2)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \hat{\mathbf{p}}_{m}^{(b)}=f_{m}\left(\mathbf{x}^{(b)}\right), \hat{\mathbf{p}}_{m}^{(b)}=\left[\hat{t}_{B, m}^{(b)}, \hat{t}_{X, m}^{(b)}\right], \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>where <italic>f<sub>m</sub></italic>(·) denotes the <italic>m</italic>-th feature-point localization strategy, and <inline-formula><tex-math id="M4">$$ \hat{t}_{B, m}^{(b)}$$</tex-math></inline-formula> and <inline-formula><tex-math id="M4">$$\hat{t}_{X, m}^{(b)}$$</tex-math></inline-formula> represent the predicted temporal positions of the B point and X point, respectively, in the <italic>b</italic>-th cardiac cycle. In general, the B point corresponds to an event temporally adjacent to AVO and is used to characterize the onset of ventricular ejection, whereas the X point corresponds to an event adjacent to AVC and is used to characterize the termination of ejection. Together, these two points define the key systolic interval most closely associated with the ejection process and directly determine the accuracy and stability of ICG-derived parameters, including LVET, SV, and CO. Accordingly, in this study, localization errors of the B and X points are regarded as one of the major sources of downstream error in hemodynamic parameter estimation.</p>
        <p>To avoid conceptual ambiguity, it should be clarified that the B and X points referred to in this study primarily represent the feature-point locations identified on the ICG waveform by the proposed model or rule-based method. Both supervised training and performance evaluation used the publicly available AVO and AVC event timestamps as reference labels in the HeartCycle dataset. Therefore, the algorithm-derived B and X points were not treated as independently annotated ground-truth labels. Instead, the AVO and AVC reference labels were used as unified reference standards for the onset and termination of ventricular ejection, respectively, against which ICG feature-point localization and LVET estimation were evaluated. It should be noted that there may be systematic deviations between the B/X annotation and the actual valve event. For example, changes in waveform morphology may lead to different algorithms having inconsistent definitions of point B. Accordingly, point-wise localization error alone is insufficient to fully capture the practical value of a given method. A more meaningful evaluation is to examine whether such localization errors materially affect the final hemodynamic estimates, which is precisely why this study places greater emphasis on parameter-level performance. After obtaining <inline-formula><tex-math id="M4">$$ \hat{t}_{B, m}^{(b)}$$</tex-math></inline-formula> and <inline-formula><tex-math id="M4">$$\hat{t}_{X, m}^{(b)}$$</tex-math></inline-formula>, a unified deterministic post-processing procedure was applied to map the localized feature points to beat-to-beat hemodynamic parameters. First, the left ventricular ejection time is defined as</p>
		
       <p><disp-formula> <label>(3)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} \hat{\mathrm{LVET}_{m}^{(b)}}=\hat{t}_{X, m}^{(b)}-\hat{t}_{B, m}^{(b)} . \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Subsequently, <inline-formula><tex-math id="M4">$${\mathrm{\hat{LVET}}_{m}^{(b)}}$$</tex-math></inline-formula> is combined with the corresponding ICG-derived morphological descriptor <inline-formula><tex-math id="M4">$$\phi_{\mathrm{ICG}}^{(b)}$$</tex-math></inline-formula> and fed into a predefined parametric function <italic>g</italic>(·) to yield the beat-to-beat estimate of SV:</p>
        <p><disp-formula> <label>(4)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} \hat{\mathrm{SV}}_{m}^{(b)}=g\left(\hat{\mathrm{LVET}}_{m}^{(b)}, \phi_{\mathrm{ICG}}^{(b)}\right).   \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Finally, CO is computed from the corresponding heart rate <italic>HR</italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> for that beat:</p>
		
       <p><disp-formula> <label>(5)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} \hat{\mathrm{CO}}_{m}^{(b)}=\frac{\hat{\mathrm{SV}}_{m}^{(b)} \times \mathrm{HR}^{(b)}}{1000} . \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Specifically, the ICG morphological descriptor for each cardiac beat was defined as the absolute peak amplitude of the ICG signal at the systolic C point. Its detailed definition and parameter-calibration procedure are provided in Section “Parameter derivation and evaluation metrics”. To ensure a fair comparison across methods, all experiments employed the same parameter derivation function, g(·), thereby minimizing the influence of subsequent parameter-conversion procedures. Accordingly, performance differences between methods were intended to primarily reflect their ability to localize ICG fiducial points.</p>
        <sec id="sec2-3-1">
          <title>Conventional rule-based localization method</title>
          <p>The conventional approach follows a rule-based ICG feature-point localization pipeline. Cardiac cycles are segmented at the beat level using the ECG R peak as the temporal anchor. The overall workflow is illustrated in <xref ref-type="fig" rid="fig4">Figure 4</xref>.</p>
          <fig id="fig4" position="float">
            <label>Figure 4</label>
            <caption>
              <p>Workflow of the conventional rule-based feature-point localization method.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.4.jpg" />
          </fig>
          <p>Within each cardiac cycle, the systolic dominant peak (C point) is first identified on the ICG waveform to establish the principal systolic morphology. Subsequently, the feature points associated with AVO and closure - the B point and X point, respectively - are searched within predefined intervals before and after the C point. The LVET is then obtained as the temporal difference between B and X, and is further used to derive SV and CO. To reduce the temporal ambiguity associated with localization based solely on ICG morphology, we design two conventional localization strategies. The first uses only the rhythm information provided by ECG to constrain the search windows for the B and X points. The second further incorporates the mechanical-event timing priors provided by S1 and S2 in the PCG to narrow the candidate search intervals for the B and X points, without directly treating S1 and S2 as the corresponding B and X points. These strategies reduce uncertainty in fiducial point searching and improve localization stability and robustness under noise interference, interindividual variability, and changes in waveform morphology.</p>
        </sec>
        <sec id="sec2-3-2">
          <title>ECG-only method</title>
          <p>As a baseline for conventional feature-point localization, this approach relies exclusively on ECG-derived electrical activity as the physiological anchor to constrain the search space for ICG features. Specifically, R peaks are first detected from the ECG using a modified Pan-Tompkins algorithm<sup>[<xref ref-type="bibr" rid="B10">10</xref>]</sup>, and the continuous recordings are segmented into individual cardiac cycles based on these fiducial points. Feature-point localization is then performed in a sequential manner within each beat.</p>
          <p>The systolic dominant peak (C point), corresponding to the moment of maximal left ventricular ejection velocity, is first identified. The algorithm searches for the global maximum of the ICG waveform within a predefined systolic window following the R peak. Accurate localization of the C point establishes a reliable temporal and amplitude reference for subsequent steps. The B point, physiologically associated with AVO and marking the onset of ejection, is then determined. Given inter-individual variability in the pre-ejection period and the frequent distortion of the ICG upstroke due to baseline drift or superimposed noise, the algorithm first searches for a local minimum on the upstroke within an adaptive window between the R peak and the C point, selecting the trough closest to the leading edge of the C point. However, because morphological variability often renders such minima indistinct or absent, a robust threshold-crossing fallback strategy is introduced. When no stable minimum is detected, a fixed proportion of the C-point amplitude is used as an adaptive threshold, and the first upward crossing of this threshold along the rising limb is taken as a surrogate for the B point.</p>
          <p>Finally, the X point, corresponding to AVC and the termination of ejection, is localized. In a typical ICG waveform, this point appears as a pronounced dicrotic notch spanning late systole to early diastole. The algorithm identifies the absolute minimum within the interval between the C point and the subsequent R peak. To prevent misclassification of later diastolic features - such as the O wave (associated with mitral valve opening) - or structures from the next cardiac cycle, a strict physiological constraint is imposed on the upper bound of the search window, thereby effectively reducing cross-cycle localization drift.</p>
        </sec>
        <sec id="sec2-3-3">
          <title>ECG + PCG-assisted method</title>
          <p>Because localization based solely on single-modality ICG morphology is highly susceptible to temporal ambiguity under complex conditions, the ECG + PCG-assisted method incorporates PCG as prior information related to cardiac valve mechanical events. Heart sounds capture acoustic manifestations of valve closure and hemodynamic impact. The first heart sound, S1, occurs near the isovolumetric contraction phase and is temporally associated with the B point, corresponding to AVO. The second heart sound, S2, is primarily generated by the closure of the aortic and pulmonary valves and therefore provides an important mechanical timing cue for anchoring the X point, corresponding to AVC. To achieve temporal coordination across modalities, the PCG signal is first subjected to tailored preprocessing. After bandpass filtering to suppress respiratory sounds and low-frequency artifacts, the signal amplitude is converted to its absolute value and smoothed using a low-pass filter, or the Hilbert transform is applied to extract the heart sound energy envelope. Within each R-R interval, an adaptive peak detection algorithm is then used to identify the energy centers of S1 and S2. Spurious peaks caused by environmental noise are further removed using energy thresholds and empirically defined temporal windows.</p>
          <p>The PCG signal is first processed using a 20-200 Hz fourth-order Butterworth bandpass filter. After taking the absolute amplitude, the signal is downsampled to approximately 1,000 Hz, and a 15 Hz fourth-order low-pass filter is applied to obtain a normalized energy envelope. The amplitude threshold for high-confidence heart sound peaks is defined as the envelope median plus six times the median absolute deviation. The lower bound of the S1 search window is set to 20 ms after the Q point. When the Q point is unavailable, the lower bound is set to 20 ms after the R peak. The upper bound is defined as min(200 ms after the R peak, 300 ms before the next R peak). The S2 search window extends from max(150 ms after S1, 200 ms after the R peak) to min(50 ms before the next R peak, 650 ms after the R peak, 0.75RR after the R peak). S1 or S2 is marked as a reliable event, and the corresponding temporal constraint is activated, only when the candidate peak exceeds the amplitude threshold. Otherwise, the constraint associated with that heart sound event is removed, and the algorithm reverts to the ECG-only search rules. When S1 is reliable, the B-point search window is defined from max(10 ms after the R peak, 10 ms after the Q point, 5 ms after S1) to min(350 ms after the R peak, 20 ms before the C point, 180 ms after S1), with the center of the temporal proximity score set to 60 ms after S1. For B-point candidate minima, the minimum separation is set to 30 ms, the prominence threshold is set to 0.04 times the standard deviation of the candidate interval, and the width of the temporal proximity score is set to 50 ms. If no valid local minimum is identified, the algorithm retains the threshold-crossing rule based on 15% of the C-point amplitude to detect the B point on the ascending limb. When S2 is reliable, the X-point search window is defined from max(50 ms after the C point, 80 ms before S2) to min(50 ms before the next R peak, 0.90RR after the R peak, 900 ms after the R peak, 160 ms after S2), with the center of the temporal proximity score set to 20 ms after S2. For X-point candidate minima, the minimum separation is set to 30 ms, the prominence threshold is set to 0.03 times the standard deviation of the candidate interval, and the width of the temporal proximity score is set to 80 ms. If no local minimum satisfies these criteria, the absolute minimum within the constrained interval is selected. When S1 or S2 is unreliable, localization of the corresponding B or X point automatically reverts to the wide-window ECG-only rules.</p>
        </sec>
      </sec>
      <sec id="sec2-4">
        <title>Multimodal event localization based on deep learning</title>
        <p>This study reformulates the ICG fiducial point localization problem as a deep learning-based sequence modeling task. The proposed framework automatically learns the temporal patterns associated with AVO and closure events from multimodal signals, including ECG, ICG, and PCG, in a data-driven manner, thereby improving the stability of fiducial point localization under complex waveform conditions. The overall workflow is illustrated in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The primary advantage of the deep learning approach lies in its ability to extract high-level representations directly from raw signals and to integrate complementary information across modalities within an end-to-end architecture, thereby reducing reliance on handcrafted rules and prior assumptions inherent to conventional methods. In this section, the proposed model is described in detail, including its input representation, network architecture, training strategy, and inference procedure.</p>
        <fig id="fig5" position="float">
          <label>Figure 5</label>
          <caption>
            <p>Workflow of the proposed deep learning-based method.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.5.jpg" />
        </fig>
        <p>For each cardiac cycle, the ECG R peak was used as the temporal anchor, and multimodal signal segments from 0.20 s before to 0.80 s after the R peak were extracted to construct a fixed 1.00 s beat-level analysis window. This window covers the main physiological process from electrical activation to the end of mechanical systole and includes the typical temporal range of the B and X points. To ensure temporal consistency across modalities and reduce computational complexity, all signals were resampled to 200 Hz; therefore, each beat-level input sample contained 200 time points. The ECG signal was z-score standardized beat by beat after resampling and used as the electrical timing reference. The ICG signal was represented by its first derivative, computed as the temporal gradient of the resampled impedance signal and then z-score standardized within each beat, to emphasize rapid hemodynamic waveform changes. For PCG, when the original sampling rate was high, the signal was first resampled to 4,000 Hz, then filtered using a fourth-order Butterworth 20-800 Hz band-pass filter, transformed into an analytic-signal amplitude envelope using the Hilbert transform, resampled to 200 Hz, and standardized within each beat. Compared with the raw high-frequency PCG waveform, the envelope representation preserves key valvular acoustic events such as S1 and S2 while reducing training instability caused by high-frequency oscillations.</p>
        <p>To fairly evaluate the contribution of PCG, two input configurations were evaluated: a trimodal setting (ECG + ICG + PCG) and a bimodal setting (ECG + ICG, with the PCG channel set to zero). The network architecture was kept identical between settings so that performance differences could be attributed to information content rather than model capacity. The proposed deep learning model consists of four core components: modality-specific encoders, a multimodal fusion and contextual modeling module, an event-query localization head, and an auxiliary regression constraint module. The model was designed to capture both local waveform morphology and global temporal dependencies, while introducing physiological consistency constraints into the localization task. To avoid mixing the different statistical properties of the modalities too early, an independent one-dimensional convolutional encoder was assigned to each modality. Each encoder maps a single-channel sequence to a high-dimensional feature representation:</p>
       <p><disp-formula> <label>(6)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \mathrm{F}_{m}=E_{m}\left(\mathrm{x}_{m}\right) \in \mathbb{R}^{B \times C \times T} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Each encoder consists of an initial stem convolutional layer with kernel size 7, followed by stacked residual dilated convolutional blocks. The residual blocks use convolutions with kernel size 5 and a dilation-rate sequence of [1, 2, 4, 8, 16, 8, 4, 2], forming a multi-scale receptive field that captures millisecond-level local waveform details and broader morphological context over hundreds of milliseconds. The three modality-specific feature representations are then concatenated along the channel dimension and projected into a shared latent space with hidden dimension d = 64 through a convolutional layer, yielding the fused temporal representation:</p>
        <p><disp-formula> <label>(7)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \mathrm{U}=\operatorname{Convld}_{1 \times 1}\left(\left[\mathrm{~F}_{\mathrm{ECG}} ; \mathrm{F}_{\mathrm{ICG}} ; \mathrm{F}_{\mathrm{PCG}}\right]\right) \in \mathbb{R}^{B \times d \times T}   \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>The fused representation is transposed into a temporal-token sequence and combined with sinusoidal positional encoding to inject absolute timing information. The resulting sequence is passed to a module composed of two Transformer encoder layers, whose self-attention mechanism captures long-range dependencies across the entire cardiac window:</p>
       <p><disp-formula> <label>(8)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \mathrm{H}=\operatorname{Transformer}(\mathrm{S}+\mathrm{P}) \in \mathbb{R}^{B \times T \times d} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>In the event localization head, an event-query retrieval mechanism was adopted to jointly exploit the complementary temporal information provided by ECG, ICG, and PCG signals. Specifically, the model defines two learnable event query vectors, <inline-formula><tex-math id="M4">$$\mathrm{q}_{1}, \mathrm{q}_{2} \in \mathbb{R}^{d}$$</tex-math></inline-formula>, corresponding to the AVO and AVC events, respectively. The hidden dimension of the fused temporal representation was set to d = 64 , and the Transformer output sequence is denoted as <inline-formula><tex-math id="M4">$$\mathrm{H} \in \mathbb{R}^{B \times T \times 64}$$</tex-math></inline-formula>, where B is the batch size and T is the number of time steps. The two event query vectors are stacked row-wise to form <inline-formula><tex-math id="M4">$$ \mathrm{E} \in \mathbb{R}^{2 \times 64}$$</tex-math></inline-formula>. The linear projection matrices W<italic><sub>q</sub></italic> and W<italic><sub>k</sub></italic> are both implemented using bias-free fully connected layers, i.e., Linear(64,64), such that <inline-formula><tex-math id="M4">$$ \mathrm{W}_{q}, \mathrm{W}_{k} \in \mathbb{R}^{64 \times 64}$$</tex-math></inline-formula>. For the b-th sample, the event queries and temporal context are projected as <inline-formula><tex-math id="M4">$$\mathrm{Q}^{(b)}=\mathrm{EW}_{q}^{\mathrm{T}} \in \mathbb{R}^{2 \times 64}$$</tex-math></inline-formula> and <inline-formula><tex-math id="M4">$$\mathrm{K}^{(b)}=\mathrm{H}^{(b)} \mathrm{W}_{k}^{\mathrm{T}} \in \mathbb{R}^{T \times 64}$$</tex-math></inline-formula>, respectively. Equivalently, the query vector for the e-th event type and the key vector at the t-th time step can be expressed as <inline-formula><tex-math id="M4">$$\tilde{\mathrm{q}}_{e}=\mathrm{W}_{q} \mathrm{q}_{e}$$</tex-math></inline-formula> and <inline-formula><tex-math id="M4">$$\mathrm{k}_{t}^{(b)}=\mathrm{W}_{k}\mathrm{h}_{t}^{(b)}$$</tex-math></inline-formula>. The logits for event e at time point t are then computed using scaled dot-product attention:</p>
		
      <p><disp-formula> <label>(9)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} Z_{e}(t)=\frac{\tilde{\mathrm{q}}_{e}^{T} \mathrm{k}_{t}}{\sqrt{d}} \cdot \exp (s)+b_{e}, e \in\{\mathrm{AVO}, \mathrm{AVC}\}  \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>where <italic>s</italic> is a learnable scaling parameter and <italic>b<sub>e</sub></italic> is the event-specific bias term. This yields a logit heatmap <inline-formula><tex-math id="M4">$$Z \in \mathbb{R}^{B \times 2 \times T}$$</tex-math></inline-formula>. This design is intended to allow the model to learn adaptive event-related representations and perform retrieval-based matching on the fused temporal representations, thereby providing an adaptive event-localization strategy for addressing inter-individual variability and noise interference.</p>
        <p>In addition to the primary localization task, the model incorporates an auxiliary regression branch to impose physiological consistency constraints. First, an additive temporal attention module is applied to <italic>H</italic> obtain a global context vector, <inline-formula><tex-math id="M4">$$c \in \mathbb{R}^{B \times d}$$</tex-math></inline-formula>. Next, the event logits are transformed by a softmax operation into temporal attention weights <italic>w<sub>e</sub></italic>(<italic>t</italic>), which are then used to compute event-specific context vectors, <inline-formula><tex-math id="M4">$$c_{e} \in \mathbb{R}^{B \times d}$$</tex-math></inline-formula>, through weighted summation over <italic>H</italic>. The concatenated representation [<italic>c</italic>, <italic>c<sub>AVO</sub></italic>, <italic>c<sub>AVC</sub></italic>, <italic>HR</italic>] is subsequently fed into a two-layer multilayer perceptron (MLP) to predict CO and an auxiliary estimate of Δ<sub>LVET</sub>. This regression design, combining global context and event-specific context, allows the regression and localization tasks to share a common temporal representation while explicitly focusing on event-relevant segments, thereby providing an additional physiological plausibility constraint for event localization. Model training is performed using a multi-task joint optimization strategy, in which the overall loss consists of both a heatmap-based localization loss and a physiological parameter regression loss. For the localization task, after mapping the reference event times of AVO and AVC, denoted by c, to the corresponding window indices, one-dimensional Gaussian soft labels were constructed as the supervision targets:</p>
        <p><disp-formula> <label>(10)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  y(t)=\exp \left(-\frac{(t-c)^{2}}{2 \sigma^{2}}\right)  \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>where σ corresponds to a temporal tolerance of approximately 18 ms (converted into sample points according to the sampling rate), which helps mitigate annotation jitter and provides more stable gradients during training. The heatmap localization loss is implemented using focal binary cross-entropy with logits to address the extreme class imbalance along the temporal axis between positive samples (event locations) and negative samples (non-event time points).</p>
        <p>For the regression task, predictions of CO and Δ<sub>LVET</sub> were optimized using the Smooth L1 (Huber) loss, with missing labels masked during training. To further enforce internal physiological consistency, the model derived the predicted SV from the predicted mean <inline-formula><tex-math id="M4">$$\bar{CO}$$</tex-math></inline-formula> and the known heart rate <italic>HR</italic> according to <inline-formula><tex-math id="M4">$$\bar{SV}$$</tex-math></inline-formula> = <inline-formula><tex-math id="M4">$$\bar{CO}$$</tex-math></inline-formula>·1000/<italic>HR</italic>, and the discrepancy between <inline-formula><tex-math id="M4">$$\bar{SV}$$</tex-math></inline-formula> and the ground-truth <italic>SV</italic> was incorporated as an additional consistency regularization term. The total loss was defined as a weighted sum of all components, with fixed weights throughout the experiments. The total loss consisted of the event heatmap localization loss, CO regression loss, SV consistency constraint loss, and auxiliary LVET regression loss, with their respective weights fixed at 1.00, 0.20, 0.05, and 0.20. A focal binary cross-entropy loss with logits was used for AVO and AVC event heatmap localization, with the focusing parameter γ set to 2.0 and the class-balancing parameter α set to 0.25. Smooth L1, also known as Huber loss, was used for all regression terms related to CO, SV, and LVET. All deep learning models were trained using the AdamW optimizer with an initial learning rate of <InlineParagraph>3.0 × 10<sup>-4</sup>,</InlineParagraph> a weight decay coefficient of 1.0 × 10<sup>-2</sup>, a batch size of 32, a fixed random seed of 42, and a maximum of 50 epochs. To improve training stability, gradient clipping is applied after each backward pass, with the maximum gradient norm fixed at 1.0.</p>
        <p>During inference, the heart rate used to calculate the RR interval is constrained to 30-220 beats/min. The AVO heatmap decoding window extends from 15 ms after the R peak to min(250 ms, 0.55RR) after the R peak. The lower bound of the AVC decoding window is set to 80 ms after AVO, and the upper bound is set to min(750 ms, 0.85RR) after AVO. Heatmap logits outside these intervals are masked, and the event positions from the heatmap branch are calculated within the constrained intervals using soft-argmax with a temperature parameter of τ = 0.35. The LVET predicted by the auxiliary regression branch is constrained to 120-650 ms and is used to construct a second AVC candidate, defined as the AVO position plus the auxiliary LVET. This candidate is further constrained to the same physiological AVC interval used by the heatmap branch. The AVC heatmap confidence is defined as the maximum probability obtained after temperature-based normalization with τ = 0.35 within the constrained AVC interval. The final AVC position is obtained by fusion according to AVC<sub>final</sub> = (1 - <italic>w</italic>)AVC<sub>heatmap</sub> + <italic>w</italic>AVC<sub>aux</sub>. When the heatmap confidence is below 0.18, the fusion weight w for the auxiliary LVET pathway is set to 0.45. When the confidence is at least 0.18, w is set to 0.20. The corresponding weights of the heatmap branch are therefore 0.55 and 0.80, respectively. These search ranges, thresholds, and fusion weights remain fixed across all training and test folds, and neither the validation set nor the test set is used to determine these values.</p>
        <p>The primary hemodynamic parameter estimation experiment uses a fixed subject-disjoint data split. The training set includes CH08 and CH09, with 152 candidate beat segments in total, comprising 52 segments from CH08 and 100 from CH09. The validation set includes CH10 with 59 candidate beat segments, while the test set includes CH07 with 70 candidate beat segments. After filtering for complete reference labels, 61 valid test beats with reference values for LVET, SV, and CO are retained for unified beat-level performance evaluation in the main experiment. No subject overlap exists among the training, validation, and test sets. These 61 beats are used only to report beat-level performance on the fixed test set and are not treated as independent samples in the subsequent paired statistical analysis involving 17 subjects. Subject-level inference is conducted separately using independent test results obtained when each subject is completely held out in the 17-fold leave-one-subject-out (LOSO) evaluation.</p>
        <p>To evaluate computational efficiency, the forward inference time and parameter number of the Proposed model are measured on a Windows 10 workstation. The evaluation environment includes an Intel Core i7-14650HX CPU, an NVIDIA GeForce RTX 4060 Laptop GPU, Python 3.11.14, and PyTorch 2.5.1. The Proposed model contains 395,270 parameters, and the checkpoint file size is approximately 1.65 MB. The model input is a beat-level multimodal signal segment with a length of 1 s, consisting of 200 temporal points and three channels. On the CPU, the average forward inference time for a single heartbeat window is 23.72 ± 3.69 ms with a batch size of 1, corresponding to 1.62 ms/beat with a batch size of 64. On the GPU, the average inference time is 6.24 ± 1.13 ms for batch size 1 and 0.15 ms/beat for batch size 64. These results indicate that the proposed model has a low computational burden and can support beat-level offline analysis and near-real-time processing.</p>
        <p>To further compare the performance of different learning-based models, three baseline models, including CNN-only, BiLSTM, and ICG-only, are established using the same beat-level data partitioning, input modalities, preprocessing procedures, and evaluation metrics as the main experiment. The CNN-only baseline employs conventional one-dimensional residual convolution for feature extraction and heatmap localization without incorporating the event-token attention mechanism. The BiLSTM baseline uses bidirectional LSTM networks to model temporal dependencies and estimates LVET only from AVO/AVC heatmap localization results without an additional direct LVET regression head. Both learning-based baselines use ECG, ICG (dZ/dt), and PCG envelope as multimodal inputs and follow the same training epochs, validation-based model selection strategy, and test evaluation procedure. In addition, an ICG-only baseline model is introduced to investigate the limitations of single-channel ICG-based localization and parameter estimation. This model maintains the same data partitioning, R-peak alignment windows, preprocessing pipeline, training epochs, validation selection strategy, and test evaluation procedure as the main experiment. However, it only uses single-channel ICG signals as input and does not incorporate ECG waveform information, PCG envelope features, or the event-token attention mechanism. The ICG-only model directly predicts AVO/AVC heatmaps from the single-channel ICG representation, and LVET is calculated from the detected fiducial points. SV and CO are subsequently obtained using the same output conversion procedure as other learning-based models.</p>
      </sec>
      <sec id="sec2-5">
        <title>Parameter derivation and evaluation metrics</title>
        <p>To ensure fair comparison among localization strategies, the same hemodynamic parameter derivation pipeline was used for all methods. Regardless of whether the method was rule-based (ECG-only or ECG + PCG-assisted) or deep learning-based (DL w/o PCG or DL + PCG), each method first produce the B-point and X-point locations for every valid cardiac cycle, and LVET, SV, and CO were then calculated using the same deterministic post-processing procedure. The auxiliary regression branch in the deep learning model was used only to impose physiological consistency during training and to assist decoding for low-confidence samples during inference; it was not used directly as the source of the final reported SV and CO results.</p>
        <p>For the i-th valid cardiac cycle, if the B and X points output by the model are first represented as sample indices, they are converted to the physical time axis according to the sampling rate:</p>
        <p><disp-formula> <label>(11)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}   \hat{t}_{B}^{(b)}=\frac{\hat{n}_{B}^{(b)}}{f_{s}},~\hat{t}_{X}^{(b)}=\frac{\hat{n}_{X}^{(b)}}{f_{s}} .  \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>On this basis, left ventricular ejection time is defined as:</p>
         <p><disp-formula> <label>(12)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} \mathrm{\hat{LVET}}^{(b)}=\hat{t}_{X}^{(b)}-\hat{t}_{B}^{(b)} . \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>After obtaining <inline-formula><tex-math id="M4">$$\mathrm{\hat{LVET}}^{(b)}$$</tex-math></inline-formula>, SV and CO were calculated using a unified deterministic parameter derivation function. For the b-th valid cardiac cycle, the C-point time <italic>t<sub>C</sub></italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> was first determined from the filtered and polarity-corrected ICG waveform according to the systolic dominant-peak localization rule shared by all methods, as described in Section ECG-only method. The C point was identified within a predefined systolic search window following the R peak, with the specific search range consistent with that used in the conventional localization procedure of this study.</p>
        <p>The absolute peak amplitude of ICG at the C point was then defined as:</p>
        <p><disp-formula> <label>(13)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} A_{C}^{(b)}=\left|\frac{d Z}{d t}\left(t_{C}^{(b)}\right)\right| \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Here, <italic>A<sub>C</sub></italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> denotes the only ICG morphological descriptor used in this study for SV derivation. No waveform integral, area-based measure, second-derivative feature, template similarity measure, or other manually engineered ICG morphological indices were used. The <inline-formula><tex-math id="M4">$$\mathrm{\hat{LVET}}^{(b)}$$</tex-math></inline-formula> determined by the B and X points was incorporated as a temporal feature together with <italic>A<sub>C</sub></italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> for subsequent parameter calculation.</p>
        <p>As the HeartCycle dataset does not provide the complete subject-specific parameters required by conventional Kubicek or Sramek-Bernstein equations, including thoracic length, electrode spacing, baseline impedance, and blood resistivity, fixed constants reported in the literature were not directly adopted. Instead, the scaling coefficient <italic>K</italic><sub>SV</sub> was robustly calibrated once within each training split using the VE reference sequence provided by HeartCycle. Specifically, the VE time series was first linearly interpolated at the timing of the current R peaks to obtain beat-wise reference SV values:</p>
         <p><disp-formula> <label>(14)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned} S V_{\text {ref }}^{(b)}=\operatorname{Interp}_{\text {lin }}\left(V E, t_{R}^{(b)}\right) \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>The reference ejection time was calculated from the public AVO and AVC reference events:</p>
       <p><disp-formula> <label>(15)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  L V E T_{\mathrm{ref}}^{(b)}=t_{\mathrm{AVC}, \mathrm{ref}}^{(b)}-t_{\mathrm{AVO}, \mathrm{ref}}^{(b)} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>In the valid training-beat set <inline-formula><tex-math id="M4">$$\mathcal{D}_{\text {train }}$$</tex-math></inline-formula>, the beat-wise proportional estimate was defined as:</p>
        <p><disp-formula> <label>(16)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}   r^{(b)}=\frac{S V_{\text {ref }}^{(b)}}{A_{C}^{(b)} \cdot L V E T_{\text {ref }}^{(b)}} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>To reduce the influence of aberrant cardiac beats, outliers were first removed within the training set based on the median absolute deviation of log<italic>r</italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup>, yielding the valid set <inline-formula><tex-math id="M4">$$\mathcal{D}_{\text {train }}^{*}$$</tex-math></inline-formula>. The following quantity was then calculated:</p>
		
         <p><disp-formula> <label>(17)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  K_{\mathrm{SV}}=\operatorname{median}_{b \in \mathcal{D}_{\text {train }}^{*}}\left(r^{(b)}\right)  \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Accordingly, the predefined beat-level SV function g(∙) used in this study was:</p>
         <p><disp-formula> <label>(18)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \hat{S V}^{(b)}=g\left(\hat{L V E T}^{(b)}, \phi_{\text {ICG }}^{(b)}\right)=K_{\mathrm{SV}} \cdot A_{c}^{(b)} \cdot \hat{L V E T}^{(b)}, \phi_{\mathrm{ICG}}^{(b)}=\left\{A_{C}^{(b)}\right\} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Heart rate was obtained from adjacent R-R intervals:</p>
         <p><disp-formula> <label>(19)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  HR^{(b)}=\frac{60}{R R^{(b)}}  \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Finally, <italic>CO</italic> was calculated as:</p>
        <p><disp-formula> <label>(20)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}   \hat{C O}^{(b)}=\frac{\hat{S V}^{(b)} \cdot H R^{(b)}}{1000} \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Here, <inline-formula><tex-math id="M4">$$\hat{SV}^{(b)}$$</tex-math></inline-formula> is expressed in mL and <italic>HR</italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> in beats/min; accordingly, <inline-formula><tex-math id="M4">$$\hat{CO}^{(b)}$$</tex-math></inline-formula> is expressed in L/min. The ICG signal used to calculate <italic>A<sub>C</sub></italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> retained its physical amplitude information and was not normalized on a beat-by-beat basis. Amplitude normalization was applied only to the deep learning model inputs and was not involved in the deterministic derivation of SV or CO. The reference values for SV and CO were obtained from the VE and DC parameter sequences provided by HeartCycle, respectively, and were linearly interpolated at the timing of the current R peaks. In this study, VE and DC were used only as device-derived reference values for model training and performance evaluation, rather than being considered independent ultrasound-based or invasive gold standards.</p>
        <p>Let <inline-formula><tex-math id="M4">$$\hat{y}^{(b)}$$</tex-math></inline-formula> and <italic>y</italic><sup>(</sup><italic><sup>b</sup></italic><sup>)</sup> denote the predicted value and reference value, respectively, of parameter <italic>y</italic> for the b-th valid cardiac cycle. Given <italic>N<sub>y</sub></italic> valid samples, the mean absolute percentage error (MAPE) of parameter <italic>y</italic> is defined as:</p>
         <p><disp-formula> <label>(21)</label> <tex-math id="E1"> $$ \begin{equation}  \begin{aligned}  \mathrm{MAPE}(y)=\frac{1}{N_{y}} \sum_{b=1}^{N_{y}}\left|\frac{\hat{y}^{(b)}-y^{(b)}}{y^{(b)}}\right| \times 100 \% \end{aligned} \end{equation} $$ </tex-math>
</disp-formula></p>
        <p>Based on the above definitions, this study calculates the MAPE values for LVET, SV, and CO and evaluates ECG-only, ECG + PCG-assisted, DL w/o PCG, and DL + PCG using a unified parameter derivation procedure and consistent evaluation metric definitions. Because the number of valid beats available for parameter evaluation after filtering for complete reference labels differs across methods, cross-category comparisons between the conventional rule-based and deep learning methods are presented primarily as descriptive analyses. Method comparisons and statistical inference based on the same set of valid beats are conducted separately in the corresponding controlled experiments. MAPE was selected as the primary metric because the magnitudes of SV and CO can differ substantially among subjects; percentage error reduces the influence of scale differences on the overall statistics and better reflects the relative impact of feature-point localization errors on final functional parameter outputs.</p>
      </sec>
    </sec>
    <sec id="sec3">
      <title>RESULTS</title>
      <sec id="sec3-1">
        <title>Morphological clustering results and data distribution characteristics of ICG waveforms</title>
        <p>To quantify the intrinsic complexity of ICG waveform morphology, unsupervised clustering was first performed on 1,908 valid ICG beat segments extracted from the HeartCycle dataset. The clustering results provide an intuitive representation of the morphological diversity of ICG waveforms as well as the distributional differences across clusters. In the waveform visualizations, solid lines represent the mean morphology of each cluster, while the shaded regions indicate the mean ± one standard deviation (±SD), thereby illustrating intra-class variability and the extent of feature overlap between different clusters. The corresponding visualization results are shown in <xref ref-type="fig" rid="fig6">Figure 6</xref>.</p>
        <fig id="fig6" position="float">
          <label>Figure 6</label>
          <caption>
            <p>ICG waveform morphology of the HeartCycle dataset under the K = 5 clustering scheme. A total of N = 1,908 valid ICG beat segments are included. (A-E) correspond to Cluster 0-4, with sample sizes of <italic>n</italic> = 297, 340, 526, 612, and 133, respectively. The solid line represents the mean normalized ICG waveform of each cluster, and the shaded region indicates the mean ± SD.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.6.jpg" />
        </fig>
        <p>The statistical results are summarized in <xref ref-type="fig" rid="fig7">Figure 7</xref>. A total of 1,908 valid ICG segments from the HeartCycle dataset were included in the clustering analysis. For K = 2, the two clusters contained 625 and 1,283 segments, respectively. For K = 3, the cluster sizes were 818, 484, and 606 segments. For K = 4, the distribution further split into 713, 670, 430, and 95 segments. For K = 5, the cluster sizes were 297, 340, 526, 612, and 133 segments. Across both coarse- and fine-grained clustering settings, pronounced class imbalance was consistently observed. These results indicate substantial morphological heterogeneity of ICG waveforms across cardiac cycles, accompanied by significant inter-class overlap. As clustering granularity increases, both boundary ambiguity between clusters and imbalance in class distribution become more evident. These findings suggest that conventional feature-point localization methods based on single-modality local morphological rules are unlikely to generalize robustly across diverse waveform patterns.</p>
        <fig id="fig7" position="float">
          <label>Figure 7</label>
          <caption>
            <p>Cluster sample size distribution of the HeartCycle dataset under different numbers of clusters K. A total of N = 1,908 valid ICG beat segments are included. (A-D) correspond to K = 2, K = 3, K = 4, and K = 5, respectively. The bar height and the values displayed above each bar indicate the number of beat segments contained in the corresponding cluster.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.7.jpg" />
        </fig>
        <p>To further justify the choice of cluster number, the validity and stability of K = 2-5 clustering solutions were evaluated using internal validity metrics and repeated random seeds. K = 2 achieved the best global geometric separation according to the Silhouette, Calinski-Harabasz, and Davies-Bouldin indices, suggesting a coarse two-pattern structure in the ICG waveforms. However, K = 3 showed the highest cross-seed stability (ARI, 0.777 ± 0.145; NMI, 0.796 ± 0.116) while maintaining an interpretable minimum cluster proportion of 24.916% ± 0.368% without extremely small clusters. In contrast, K = 4 and K = 5 showed smaller minimum-cluster proportions and a higher risk of over-partitioning. Therefore, K = 3 was considered more suitable for subsequent morphology-stratified analysis when geometric separation, stability, and cluster-size interpretability were considered jointly. The quantitative results are summarized in <xref ref-type="table" rid="t1">Table 1</xref> and <xref ref-type="fig" rid="fig8">Figure 8</xref>.</p>
        <fig id="fig8" position="float">
          <label>Figure 8</label>
          <caption>
            <p>Validity and robustness analysis of ICG waveform clustering results under different numbers of clusters. A total of N = 1,908 valid ICG beat segments are included. (A) Silhouette score for K = 2-5. (B) Normalized trends of multiple clustering validity metrics. (C) Label consistency across random seeds. (D) Minimum cluster proportion and the 5% small-cluster risk threshold. Except for the normalized trend plot, the points represent the mean values over five repeated runs, and the error bars indicate ±1 standard deviation (mean ± SD). ARI and NMI are calculated from 10 pairwise label comparisons generated from five runs for each <italic>K</italic> value.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.8.jpg" />
        </fig>
        <table-wrap id="t1">
          <label>Table 1</label>
          <caption>
            <p>Internal validity, cluster-size balance, and cross-seed stability for K = 2-5 clustering</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Metric</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>K = 2</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>K = 3</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>K = 4</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>K = 5</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>Silhouette</td>
                <td>0.171 ± 0.044</td>
                <td>0.036 ± 0.018</td>
                <td>0.024 ± 0.029</td>
                <td>-0.003 ± 0.030</td>
              </tr>
              <tr>
                <td>Calinski-harabasz</td>
                <td>178.496 ± 7.221</td>
                <td>138.288 ± 1.533</td>
                <td>101.119 ± 5.116</td>
                <td>88.190 ± 10.572</td>
              </tr>
              <tr>
                <td>Davies-bouldin</td>
                <td>2.701 ± 0.077</td>
                <td>3.012 ± 0.016</td>
                <td>3.480 ± 0.655</td>
                <td>3.533 ± 0.385</td>
              </tr>
              <tr>
                <td>Separation ratio</td>
                <td>1.449 ± 0.050</td>
                <td>1.544 ± 0.006</td>
                <td>1.563 ± 0.102</td>
                <td>1.575 ± 0.049</td>
              </tr>
              <tr>
                <td>Minimum-cluster proportion (%)</td>
                <td>31.184 ± 6.293</td>
                <td>24.916 ± 0.368</td>
                <td>10.440 ± 10.320</td>
                <td>6.059 ± 2.761</td>
              </tr>
              <tr>
                <td>Stability ARI</td>
                <td>0.538 ± 0.350</td>
                <td>0.777 ± 0.145</td>
                <td>0.532 ± 0.084</td>
                <td>0.471 ± 0.076</td>
              </tr>
              <tr>
                <td>Stability NMI</td>
                <td>0.528 ± 0.247</td>
                <td>0.796 ± 0.116</td>
                <td>0.577 ± 0.058</td>
                <td>0.514 ± 0.061</td>
              </tr>
              <tr>
                <td>Medoid DTW distance</td>
                <td>1,772.82 ± 45.12</td>
                <td>1,549.95 ± 2.38</td>
                <td>1,533.89 ± 19.28</td>
                <td>1,496.79 ± 18.30</td>
              </tr>
              <tr>
                <td>Relative inertia</td>
                <td>1.000</td>
                <td>0.874</td>
                <td>0.865</td>
                <td>0.844</td>
              </tr>
              <tr>
                <td>Decrease from previous K (%)</td>
                <td>-</td>
                <td>12.57</td>
                <td>1.04</td>
                <td>2.42</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>ARI: adjusted rand index; NMI: normalized mutual information; DTW: dynamic time warping.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec id="sec3-2">
        <title>Morphological classification results of ICG waveforms</title>
        <p>To further validate the morphological diversity and learnability of ICG waveforms, this study performs morphology classification on the preprocessed ICG beat segments using the clustering-derived labels. For each <italic>K</italic> value, the test set contains <italic>n</italic> = 382 beat segments. In the classification tasks with K = 2, K = 3, K = 4, and K = 5, the complete MS-GBiLSTM-TA model achieves Accuracy values of 95.03%, 91.36%, 89.79%, and 76.96%, respectively. The corresponding Macro-F1 scores are 0.9432, 0.9134, 0.8507, and 0.7686, while the Weighted-F1 scores are 0.9501, 0.9135, 0.8982, and 0.7714, respectively. Detailed class-level Precision, Recall, and F1-score results are presented in <xref ref-type="table" rid="t2">Tables 2</xref>-<xref ref-type="table" rid="t5">5</xref>. As the number of clusters increases, the classification task becomes more difficult and model performance declines, indicating that finer-grained partitioning of ICG waveform morphologies leads to greater feature overlap and less distinct boundaries between classes.</p>
        <table-wrap id="t2">
          <label>Table 2</label>
          <caption>
            <p>Binary classification results for ICG waveforms in the HeartCycle dataset</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;"><bold>Class</bold></td>
                <td style="border-bottom:1;"><bold>Precision</bold></td>
                <td style="border-bottom:1;"><bold>Recall</bold></td>
                <td style="border-bottom:1;"><bold>F1</bold></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>0</td>
                <td>0.9344</td>
                <td>0.9120</td>
                <td>0.9231</td>
              </tr>
              <tr>
                <td>1</td>
                <td>0.9577</td>
                <td>0.9689</td>
                <td>0.9632</td>
              </tr>
              <tr>
                <td>Macro average</td>
                <td>0.9461</td>
                <td>0.9404</td>
                <td>0.9432</td>
              </tr>
              <tr>
                <td>Weighted average</td>
                <td>0.9501</td>
                <td>0.9503</td>
                <td>0.9501</td>
              </tr>
            </tbody>
          </table>
		  <table-wrap-foot>
            <fn>
              <p>ICG: impedance cardiography.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap id="t3">
          <label>Table 3</label>
          <caption>
            <p>Three-class classification results for ICG waveforms in the HeartCycle dataset</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;"><bold>Class</bold></td>
                <td style="border-bottom:1;"><bold>Precision</bold></td>
                <td style="border-bottom:1;"><bold>Recall</bold></td>
                <td style="border-bottom:1;"><bold>F1</bold></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>0</td>
                <td>0.9539</td>
                <td>0.8841</td>
                <td>0.9177</td>
              </tr>
              <tr>
                <td>1</td>
                <td>0.8571</td>
                <td>0.9897</td>
                <td>0.9187</td>
              </tr>
              <tr>
                <td>2</td>
                <td>0.9153</td>
                <td>0.8926</td>
                <td>0.9038</td>
              </tr>
              <tr>
                <td>Macro average</td>
                <td>0.9088</td>
                <td>0.9221</td>
                <td>0.9134</td>
              </tr>
              <tr>
                <td>Weighted average</td>
                <td>0.9171</td>
                <td>0.9136</td>
                <td>0.9135</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>ICG: impedance cardiography.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap id="t4">
          <label>Table 4</label>
          <caption>
            <p>Four-class classification results for ICG waveforms in the HeartCycle dataset</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;"><bold>Class</bold></td>
                <td style="border-bottom:1;"><bold>Precision</bold></td>
                <td style="border-bottom:1;"><bold>Recall</bold></td>
                <td style="border-bottom:1;"><bold>F1</bold></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>0</td>
                <td>0.9091</td>
                <td>0.9091</td>
                <td>0.9091</td>
              </tr>
              <tr>
                <td>1</td>
                <td>0.9030</td>
                <td>0.9030</td>
                <td>0.9030</td>
              </tr>
              <tr>
                <td>2</td>
                <td>0.9294</td>
                <td>0.9186</td>
                <td>0.9240</td>
              </tr>
              <tr>
                <td>3</td>
                <td>0.6500</td>
                <td>0.6842</td>
                <td>0.6667</td>
              </tr>
              <tr>
                <td>Macro average</td>
                <td>0.8479</td>
                <td>0.8537</td>
                <td>0.8507</td>
              </tr>
              <tr>
                <td>Weighted average</td>
                <td>0.8986</td>
                <td>0.8979</td>
                <td>0.8982</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>ICG: impedance cardiography.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <table-wrap id="t5">
          <label>Table 5</label>
          <caption>
            <p>Five-class classification results for ICG waveforms in the HeartCycle dataset</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;"><bold>Class</bold></td>
                <td style="border-bottom:1;"><bold>Precision</bold></td>
                <td style="border-bottom:1;"><bold>Recall</bold></td>
                <td style="border-bottom:1;"><bold>F1</bold></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>0</td>
                <td>0.9455</td>
                <td>0.8814</td>
                <td>0.9123</td>
              </tr>
              <tr>
                <td>1</td>
                <td>0.6024</td>
                <td>0.7353</td>
                <td>0.6623</td>
              </tr>
              <tr>
                <td>2</td>
                <td>0.7143</td>
                <td>0.6190</td>
                <td>0.6633</td>
              </tr>
              <tr>
                <td>3</td>
                <td>0.8814</td>
                <td>0.8455</td>
                <td>0.8631</td>
              </tr>
              <tr>
                <td>4</td>
                <td>0.6571</td>
                <td>0.8519</td>
                <td>0.7419</td>
              </tr>
              <tr>
                <td>Macro average</td>
                <td>0.7601</td>
                <td>0.7866</td>
                <td>0.7686</td>
              </tr>
              <tr>
                <td>Weighted average</td>
                <td>0.7798</td>
                <td>0.7696</td>
                <td>0.7714</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>ICG: impedance cardiography.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>
          <xref ref-type="fig" rid="fig9">Figure 9</xref> presents the corresponding confusion matrices. As the number of classes increases, the number of off-diagonal elements also increases, indicating more pronounced class confusion in fine-grained ICG morphology recognition. Further analysis of the class-level results reveals several difficult classes in the multiclass tasks. For example, in the four-class classification task, Class 3 achieves a Recall of 0.6842, a Precision of 0.6500, and an F1-score of 0.6667, indicating that this class remains difficult to identify accurately. This finding suggests that some ICG waveform subtypes exhibit high local morphological similarity to other classes, making it difficult for the model to learn stable and discriminative feature representations. The differences between the macro-averaged and weighted-averaged metrics further reflect the influence of class imbalance on performance evaluation. When minority classes are difficult to identify, the macro-averaged metrics decrease more markedly, indicating that the model does not perform uniformly across classes. Overall, the morphology classification results provide data-driven support for the preceding clustering analysis, showing that ICG waveforms exhibit substantial morphological diversity and class overlap under real-beat conditions. These findings indicate that rigidly matching specific mechanical events using only local morphological features from single-modal ICG can lead to temporal ambiguity and localization drift in complex waveforms. This observation provides further motivation for incorporating ECG and PCG into the subsequent multimodal fiducial point localization framework.</p>
        <fig id="fig9" position="float">
          <label>Figure 9</label>
          <caption>
            <p>Confusion matrices of ICG morphology classification results under different clustering schemes in the HeartCycle dataset. All confusion matrices are generated from the test set with <italic>n</italic> = 382 beat segments, and the values in each cell represent the number of samples assigned to the corresponding category. (A) Binary classification. (B) Three-class classification. (C) Four-class classification. (D) Five-class classification.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.9.jpg" />
        </fig>
        <p>The structural ablation experiments evaluate the contributions of multiscale convolution, bidirectional temporal modeling, and temporal attention across different ICG morphology classification tasks. All models are compared using the same training, validation, and test splits, and the test set for each <italic>K</italic> value contains <InlineParagraph><italic>n</italic> = 382</InlineParagraph> beat segments. The structural ablation results of the MS-GBiLSTM-TA classification model are presented in <xref ref-type="table" rid="t6">Table 6</xref>. In the binary classification task with K = 2, the CNN-only model achieves the highest Accuracy, Macro-F1, and Weighted-F1. For K = 3, the complete MS-GBiLSTM-TA model performs best across all three metrics. For K = 4, the complete model achieves the highest Accuracy and Weighted-F1, whereas the BiLSTM-only model obtains a higher Macro-F1. For K = 5, the CNN + BiLSTM model achieves the highest Accuracy and Weighted-F1, while the complete model yields the highest Macro-F1.</p>
        <table-wrap id="t6">
          <label>Table 6</label>
          <caption>
            <p>Structural ablation results of the MS-GBiLSTM-TA classification model</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>K</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Model</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>n</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Accuracy</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Macro-F1</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Weighted-F1</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td rowspan="4">2</td>
                <td>CNN-only</td>
                <td rowspan="16">382</td>
                <td>0.9555</td>
                <td>0.9496</td>
                <td>0.9555</td>
              </tr>
              <tr>
                <td>BiLSTM-only</td>
                <td>0.9188</td>
                <td>0.9080</td>
                <td>0.9189</td>
              </tr>
              <tr>
                <td>CNN + BiLSTM</td>
                <td>0.9450</td>
                <td>0.9366</td>
                <td>0.9446</td>
              </tr>
              <tr>
                <td>MS-GBiLSTM-TA</td>
                <td>0.9503</td>
                <td>0.9432</td>
                <td>0.9501</td>
              </tr>
              <tr>
                <td rowspan="4">3</td>
                <td>CNN-only</td>
                <td>0.8665</td>
                <td>0.8718</td>
                <td>0.8657</td>
              </tr>
              <tr>
                <td>BiLSTM-only</td>
                <td>0.9058</td>
                <td>0.9072</td>
                <td>0.9053</td>
              </tr>
              <tr>
                <td>CNN + BiLSTM</td>
                <td>0.8717</td>
                <td>0.8711</td>
                <td>0.8713</td>
              </tr>
              <tr>
                <td>MS-GBiLSTM-TA</td>
                <td>0.9136</td>
                <td>0.9134</td>
                <td>0.9135</td>
              </tr>
              <tr>
                <td rowspan="4">4</td>
                <td>CNN-only</td>
                <td>0.8325</td>
                <td>0.7956</td>
                <td>0.8347</td>
              </tr>
              <tr>
                <td>BiLSTM-only</td>
                <td>0.8927</td>
                <td>0.8597</td>
                <td>0.8946</td>
              </tr>
              <tr>
                <td>CNN + BiLSTM</td>
                <td>0.8848</td>
                <td>0.8465</td>
                <td>0.8886</td>
              </tr>
              <tr>
                <td>MS-GBiLSTM-TA</td>
                <td>0.8979</td>
                <td>0.8507</td>
                <td>0.8982</td>
              </tr>
              <tr>
                <td rowspan="4">5</td>
                <td>CNN-only</td>
                <td>0.7120</td>
                <td>0.6954</td>
                <td>0.7161</td>
              </tr>
              <tr>
                <td>BiLSTM-only</td>
                <td>0.7382</td>
                <td>0.7383</td>
                <td>0.7411</td>
              </tr>
              <tr>
                <td>CNN + BiLSTM</td>
                <td>0.7775</td>
                <td>0.7595</td>
                <td>0.7783</td>
              </tr>
              <tr>
                <td>MS-GBiLSTM-TA</td>
                <td>0.7696</td>
                <td>0.7686</td>
                <td>0.7714</td>
              </tr>
            </tbody>
          </table>
		 </table-wrap>
      </sec>
      <sec id="sec3-3">
        <title>ICG feature-point localization and hemodynamic parameter estimation results</title>
        <p>The preceding clustering and classification analyses demonstrate that ICG waveforms exhibit substantial morphological variability and feature overlap. This implies that relying solely on local morphological characteristics from single-modality ICG to match specific mechanical events is prone to temporal ambiguity and localization drift, which can further propagate as cumulative errors in beat-to-beat hemodynamic parameter estimation. Therefore, single-modality local morphological analysis still faces challenges in achieving stable event localization under complex waveform conditions. Motivated by these findings, we further compared the performance of conventional rule-based methods and multimodal deep learning approaches in both key feature-point localization and downstream hemodynamic parameter estimation.</p>
        <p>The comparative methods fall into two categories. The conventional rule-based approaches include the ECG-only method, which relies exclusively on ECG for temporal anchoring, and the ECG + PCG-assisted method, which further incorporates soft constraints derived from PCG heart sound events. The deep learning approaches include deep learning (DL) w/o PCG, which excludes PCG input, and DL + PCG, which explicitly integrates PCG into the multimodal framework. Error statistics were computed on the basis of the relative error for each valid beat and then aggregated as the MAPE for comparison. For overall reference, <xref ref-type="table" rid="t7">Table 7</xref> summarizes the mean MAPE (%) of all methods for LVET, SV, and CO.</p>
        <table-wrap id="t7">
          <label>Table 7</label>
          <caption>
            <p>Comparison of mean absolute percentage error (MAPE, %) across different methods</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Methods</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>LVET</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>SV</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>CO</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>ECG-only</td>
                <td>15.45</td>
                <td>14.67</td>
                <td>18.35</td>
              </tr>
              <tr>
                <td>ECG +PCG-assisted</td>
                <td>14.84</td>
                <td>13.89</td>
                <td>17.50</td>
              </tr>
              <tr>
                <td>DL w/o PCG</td>
                <td>21.11</td>
                <td>8.31</td>
                <td>9.00</td>
              </tr>
              <tr>
                <td>DL + PCG</td>
                <td>16.04</td>
                <td>6.51</td>
                <td>7.35</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>MAPE: Mean absolute percentage error; ECG: electrocardiography; PCG: phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output.</p>
            </fn>
          </table-wrap-foot>
		 </table-wrap>
        <p>To address statistical uncertainty, this study retains the beat-level mean MAPE reported in the original manuscript as the primary result and additionally reports the mean ± SD, median [IQR], and 95% CI of the error distributions for each method, as shown in <xref ref-type="table" rid="t8">Table 8</xref>. The beat-level MAPE statistics in <xref ref-type="table" rid="t8">Table 8</xref> describe the per-beat error performance and dispersion of each method within its corresponding evaluation cohort, whereas statistical inference for between-method comparisons is conducted at the subject level.</p>
        <table-wrap id="t8">
          <label>Table 8</label>
          <caption>
            <p>Descriptive statistics of beat-level MAPE for different methods (%)</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Method</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Metric</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Number of valid beat segments</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Mean ± SD</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Median [IQR]</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>95%CI</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td rowspan="3">ECG-only</td>
                <td>LVET</td>
                <td rowspan="6">92</td>
                <td>15.45 ± 6.65</td>
                <td>15.25 <break />[10.88,19.01]</td>
                <td>14.09-16.81</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>14.67 ± 6.74</td>
                <td>14.53 <break />[9.68, 17.53]</td>
                <td>13.30-16.05</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>18.35 ± 8.44</td>
                <td>16.62 <break />[12.54,23.92]</td>
                <td>16.62-20.07</td>
              </tr>
              <tr>
                <td rowspan="3">ECG + PCG-assisted</td>
                <td>LVET</td>
                <td>14.84 ± 6.75</td>
                <td>14.99 <break />[10.61,18.94]</td>
                <td>13.46-16.22</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>13.89 ± 6.52</td>
                <td>13.13 <break />[9.32, 16.28]</td>
                <td>12.56-15.22</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>17.50 ± 7.76</td>
                <td>16.42<break />[12.77,20.79]</td>
                <td>15.92-19.09</td>
              </tr>
              <tr>
                <td rowspan="3">DL w/o PCG</td>
                <td>LVET</td>
                <td rowspan="6">61</td>
                <td>21.11 ± 13.81</td>
                <td>16.81<break />[11.21,33.23]</td>
                <td>17.64-24.57</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>8.31 ± 5.10</td>
                <td>7.30<break />[4.42,11.85]</td>
                <td>7.03-9.58</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>9.00 ± 5.37</td>
                <td>9.20 <break />[5.65,12.59]</td>
                <td>7.65-10.35</td>
              </tr>
              <tr>
                <td rowspan="3">DL + PCG</td>
                <td>LVET</td>
                <td>16.04 ± 8.31</td>
                <td>17.56 <break />[9.37,21.85]</td>
                <td>13.96-18.13</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>6.51 ± 4.91</td>
                <td>5.95 <break />[2.35, 9.51]</td>
                <td>5.28-7.75</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>7.35 ± 4.56</td>
                <td>6.43<break />[3.77,10.53]</td>
                <td>6.20-8.49</td>
              </tr>
            </tbody>
          </table>
		    <table-wrap-foot>
            <fn>
              <p>MAPE: Mean absolute percentage error; ECG: electrocardiography; PCG: phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Because each subject can contribute multiple beats, individual beats are not treated as independent observations in subject-level statistical inference. The paired analysis reported in <xref ref-type="table" rid="t9">Table 9</xref> is based on the results of 17-fold LOSO cross-validation. In each fold, one subject is completely held out as the independent test subject, one subject is used for model selection, and the remaining 15 subjects are used for model training. CH01 through CH17 serve once each as the independent test subject, and the statistical result for each subject is obtained exclusively from the fold in which that subject is completely held out. DL + PCG and DL w/o PCG use identical training, validation, and test subject assignments in every fold and are evaluated on the common test beats with complete reference values. The MAPE values for LVET, SV, and CO are first aggregated within each test subject, after which paired tests are conducted using the aggregated results from the 17 independent test subjects. For both method comparisons, ΔMAPE is calculated within each subject as the MAPE of the method without PCG minus that of the corresponding PCG-assisted method. Specifically, <InlineParagraph>ΔMAPE</InlineParagraph> is calculated as ECG-only minus ECG + PCG-assisted for the conventional rule-based comparison and as DL w/o PCG minus DL + PCG for the deep learning comparison. The ΔMAPE reported in the table is the mean of the 17 subject-specific differences; therefore, a positive value indicates that incorporating PCG reduces the estimation error. Both the paired t-test and the Wilcoxon signed-rank test use the aggregated results from the 17 subjects as paired observations. All 17 paired observations are derived from independent test results obtained when the corresponding subject is excluded from both model training and model selection, with the statistical results presented in <xref ref-type="table" rid="t9">Table 9</xref>. Because the conventional rule-based methods do not involve model training, their paired comparison is likewise aggregated at the subject level across all 17 subjects and is calculated using only the common beats for which both methods yield valid results and complete reference labels.</p>
        <table-wrap id="t9">
          <label>Table 9</label>
          <caption>
            <p>Paired statistical tests based on subject-level MAPE</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;"><bold>Comparison</bold></td>
                <td style="border-bottom:1;"><bold>Metric</bold></td>
                <td style="border-bottom:1;"><bold>Number of subjects</bold></td>
                <td style="border-bottom:1;"><bold>ΔMAPE(%)</bold></td>
                <td style="border-bottom:1;"><bold>Paired t-test P</bold></td>
                <td style="border-bottom:1;"><bold>Wilcoxon P</bold></td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td rowspan="3">ECG + PCG-assisted <italic>vs</italic>. ECG-only</td>
                <td>LVET</td>
                <td rowspan="6">17</td>
                <td>0.46</td>
                <td>0.312</td>
                <td>0.281</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>0.78</td>
                <td>0.425</td>
                <td>0.463</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>0.83</td>
                <td>0.389</td>
                <td>0.412</td>
              </tr>
              <tr>
                <td rowspan="3">DL + PCG <italic>vs</italic>. DL w/o PCG</td>
                <td>LVET</td>
                <td>0.98</td>
                <td>0.512</td>
                <td>0.488</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>2.43</td>
                <td>0.004</td>
                <td>0.007</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>2.16</td>
                <td>0.028</td>
                <td>0.035</td>
              </tr>
            </tbody>
          </table>
		   <table-wrap-foot>
            <fn>
              <p>MAPE: Mean absolute percentage error; ECG: electrocardiography; PCG: phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
        <p>Based on the independent test results from the 17-fold LOSO evaluation reported in <xref ref-type="table" rid="t9">Table 9</xref>, DL + PCG shows overall reductions in subject-level MAPE for SV and CO compared with DL w/o PCG, with both the paired t-test and the Wilcoxon signed-rank test indicating statistically significant differences. The ΔMAPE for LVET is also positive, although the difference does not reach statistical significance. For the conventional rule-based methods, ECG + PCG-assisted methods yield positive ΔMAPE values for LVET, SV, and CO compared with ECG-only, but none of the paired tests for these three metrics reach statistical significance. These results indicate that, under subject-independent LOSO evaluation, the mechanical-event timing information provided by PCG offers a clearer complementary contribution to SV and CO estimation within the deep learning framework, whereas the improvements observed in the conventional rule-based methods are primarily numerical trends.</p>
        <sec id="sec3-3-1">
          <title>Baseline performance of conventional rule-based methods and gains from PCG soft constraints</title>
          <p>Within the conventional rule-based localization framework, the ECG-only method relies solely on ECG-derived electrical activity as a temporal anchor to constrain the search windows for the B and X points. Consequently, its performance is highly dependent on the local morphological characteristics of the ICG waveform. However, as demonstrated in the preceding clustering and classification analyses, real-world ICG signals exhibit substantial morphological variability and feature overlap, often accompanied by distortions such as double peaks, flattened incisura, and baseline drift. Under such conditions, the ECG-only method is more susceptible to local morphological ambiguities, leading to localization errors that propagate into the parameter estimation stage. As shown in <xref ref-type="fig" rid="fig10">Figure 10</xref>, the resulting MAPE values for LVET, SV, and CO are 15.45%, 14.67%, and 18.35%, respectively. In contrast, the ECG + PCG-assisted method incorporates mechanical event priors derived from PCG, using S1 and S2 to impose soft constraints on the candidate intervals for the B and X points. The inclusion of PCG leads to consistent, albeit modest, improvements across all three metrics: the MAPE for LVET decreases from 15.45% to 14.84%, SV from 14.67% to 13.89%, and CO from 18.35% to 17.50%. These results indicate that even within a rule-based framework, PCG-derived mechanical priors can partially mitigate temporal ambiguity caused by ICG waveform distortion and reduce gross localization errors. Overall, while incorporating PCG provides measurable performance gains, the improvements remain limited. This suggests that constraining search windows alone can reduce extreme errors to some extent, but is insufficient to fundamentally resolve the feature-point ambiguity arising from the complex and variable morphology of ICG waveforms. A direct numerical comparison of the two conventional rule-based methods is provided in <xref ref-type="table" rid="t10">Table 10</xref>.</p>
          <fig id="fig10" position="float">
            <label>Figure 10</label>
            <caption>
              <p>Beat-level MAPE distributions of the conventional rule-based method under different auxiliary information conditions. (A and B) correspond to ECG-only and ECG + PCG-assisted, respectively. Each method is evaluated on <italic>n</italic> = 92 valid test beats. Each dot represents an individual beat; the width of the half violin indicates the kernel density distribution of MAPE values, the horizontal line represents the median, and the diamond indicates the mean.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.10.jpg" />
          </fig>
          <table-wrap id="t10">
            <label>Table 10</label>
            <caption>
              <p>Comparison of results for conventional rule-based methods</p>
            </caption>
            <table frame="hsides" rules="groups">
              <thead>
                <tr>
                  <td style="border-bottom:1;">
                    <bold>Methods</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>LVET</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>SV</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>CO</bold>
                  </td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>ECG-only</td>
                  <td>15.45</td>
                  <td>14.67</td>
                  <td>18.35</td>
                </tr>
                <tr>
                  <td>ECG + PCG-assisted</td>
                  <td>14.84</td>
                  <td>13.89</td>
                  <td>17.50</td>
                </tr>
              </tbody>
            </table>
			 <table-wrap-foot>
            <fn>
              <p>ECG: Electrocardiography; PCG: phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output.</p>
            </fn>
          </table-wrap-foot>
          </table-wrap>
        </sec>
        <sec id="sec3-3-2">
          <title>Parameter estimation performance of multimodal deep learning fusion</title>
          <p>In the deep learning framework, DL + PCG achieves relatively low errors in SV and CO estimation. As shown in <xref ref-type="fig" rid="fig11">Figure 11</xref>, the best-performing conventional rule-based method, ECG + PCG-assisted, is evaluated on <italic>n</italic> = 92 valid test beats, whereas DL + PCG is evaluated on <italic>n</italic> = 61 valid test beats with complete reference values for LVET, SV, and CO. Within their respective evaluable beat sets, ECG + PCG-assisted yields mean MAPE values of 13.89% for SV and 17.50% for CO, while DL + PCG yields corresponding values of 6.51% and 7.35%. These side-by-side results indicate that end-to-end multimodal modeling achieves lower SV and CO error levels under the current evaluation conditions. Because the two method categories are evaluated on different numbers of valid beats, this cross-category numerical comparison is presented descriptively rather than as a direct statistical comparison. The specific contribution of PCG to the deep learning model is further assessed in the subsequent ablation experiment using the same set of test beats.</p>
          <fig id="fig11" position="float">
            <label>Figure 11</label>
            <caption>
              <p>Beat-level MAPE distributions of the best-performing conventional method and the multimodal deep learning fusion method. (A and B) correspond to ECG + PCG-assisted (<italic>n</italic> = 92) and DL + PCG (<italic>n</italic> = 61), respectively. Each dot represents an individual valid beat; the width of the half violin indicates the kernel density distribution of MAPE values, the horizontal line represents the median, and the diamond indicates the mean.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.11.jpg" />
          </fig>
          <p>For the deep learning models with beat-level true and predicted outputs, additional analyses were conducted using mean absolute error (MAE), root mean square error (RMSE), Pearson correlation coefficients, and Bland-Altman agreement analysis, as presented in <xref ref-type="table" rid="t11">Table 11</xref> and <xref ref-type="fig" rid="fig12">Figure 12</xref>. These analyses were intended to complement the MAPE-based results from the perspectives of absolute error, linear correlation, and agreement, without changing the original presentation of beat-level mean MAPE as the primary outcome measure. The results showed that DL + PCG yielded lower absolute errors than DL w/o PCG for both SV and CO estimation, with MAEs of 7.12 ± 5.76 mL and 0.53 ± 0.36 L/min, respectively, and RMSEs of 9.13 mL and 0.64 L/min, respectively. Bland-Altman analysis showed biases of -6.71 mL for SV and -0.49 L/min for CO with DL + PCG, with corresponding 95% limits of agreement of -18.95 to 5.53 mL and -1.31 to 0.33 L/min, respectively. Pearson correlation coefficients were reported as supplementary metrics to reflect the linear correspondence between predicted and reference values across the dynamic range of the test set.</p>
          <fig id="fig12" position="float">
            <label>Figure 12</label>
            <caption>
              <p>Bland-Altman agreement analysis of beat-level estimation results from the deep learning model (<italic>n</italic> = 61 predicted-reference pairs in each panel). (A-C) correspond to LVET, SV, and CO, respectively. The solid line represents the mean bias, and the dashed lines indicate the 95% limits of agreement.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.12.jpg" />
          </fig>
          <table-wrap id="t11">
            <label>Table 11</label>
            <caption>
              <p>Absolute error, correlation, and agreement analysis for beat-level predictions of the deep learning models</p>
            </caption>
            <table frame="hsides" rules="groups">
              <thead>
                <tr>
                  <td style="border-bottom:1;">
                    <bold>Method</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Metric</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>n</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>MAE ± SD</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>RMSE</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Pearson r</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Bias (95% LoA)</bold>
                  </td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td rowspan="3">DL w/o PCG</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>51.79 ± 28.73 ms</td>
                  <td>59.12 ms</td>
                  <td>-0.080</td>
                  <td>51.79 (-4.52, 108.11) ms</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>8.87 ± 5.68 mL</td>
                  <td>10.51 mL</td>
                  <td>0.282</td>
                  <td>-6.62 (-22.76, 9.52) mL</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>0.64 ± 0.40 L/min</td>
                  <td>0.75 L/min</td>
                  <td>0.190</td>
                  <td>-0.48 (-1.63, 0.66) L/min</td>
                </tr>
                <tr>
                  <td rowspan="3">DL + PCG</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>43.55 ± 24.39 ms</td>
                  <td>49.81 ms</td>
                  <td>0.088</td>
                  <td>-41.90 (-95.13, 11.32) ms</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>7.12 ± 5.76 mL</td>
                  <td>9.13 mL</td>
                  <td>0.074</td>
                  <td>-6.71 (-18.95, 5.53) mL</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>0.53 ± 0.36 L/min</td>
                  <td>0.64 L/min</td>
                  <td>0.200</td>
                  <td>-0.49 (-1.31, 0.33) L/min</td>
                </tr>
              </tbody>
            </table>
			<table-wrap-foot>
            <fn>
              <p>PCG: Phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output; MAE: mean absolute error; RMSE: root mean square error.</p>
            </fn>
          </table-wrap-foot>
          </table-wrap>
          <p>Building on these results, additional comparisons with learning-based baseline models were conducted, as shown in <xref ref-type="table" rid="t12">Table 12</xref> and <xref ref-type="fig" rid="fig13">Figure 13</xref>, to exclude the possibility that the performance improvement was solely attributable to the increased capacity of a generic deep sequential model. On the same beat-level test set used in the main experiments, the CNN-only, BiLSTM, and CNN + BiLSTM baseline models were trained using identical ECG, ICG, and PCG inputs and the same signal preprocessing pipeline. Because multiple architectural differences exist between these baseline models and the Proposed model, this comparison is used to evaluate the relative performance of the complete framework rather than to quantitatively isolate the independent contribution of the event query mechanism or any individual structural component.</p>
          <fig id="fig13" position="float">
            <label>Figure 13</label>
            <caption>
              <p>Comparison of MAPE values for beat-level hemodynamic parameter estimation among different learning-based models. (A-C) correspond to LVET, SV, and CO, respectively. The bar height represents the mean MAPE calculated from <italic>n</italic> = 61 valid test beats, and the error bars indicate mean ± SD.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.13.jpg" />
          </fig>
          <table-wrap id="t12">
            <label>Table 12</label>
            <caption>
              <p>Performance comparison of learning-based baseline models on the main beat-level test set</p>
            </caption>
            <table frame="hsides" rules="groups">
              <thead>
                <tr>
                  <td style="border-bottom:1;">
                    <bold>Model</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Metric</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>n</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>MAE</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>RMSE</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>MAPE mean ± SD (%)</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Median [IQR] (%)</bold>
                  </td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td rowspan="3">CNN-only</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>59.92</td>
                  <td>65.97</td>
                  <td>22.29 ± 9.24</td>
                  <td>23.45 [13.87]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>32.08</td>
                  <td>32.75</td>
                  <td>30.23 ± 4.80</td>
                  <td>29.93 [5.92]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>2.18</td>
                  <td>2.22</td>
                  <td>30.67 ± 4.57</td>
                  <td>29.94 [6.00]</td>
                </tr>
                <tr>
                  <td rowspan="3">BiLSTM</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>120.29</td>
                  <td>123.06</td>
                  <td>47.45 ± 15.21</td>
                  <td>41.98 [23.49]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>42.55</td>
                  <td>42.97</td>
                  <td>40.17 ± 3.31</td>
                  <td>39.73 [4.91]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>2.87</td>
                  <td>2.90</td>
                  <td>40.54 ± 3.39</td>
                  <td>39.97 [4.27]</td>
                </tr>
                <tr>
                  <td rowspan="3">CNN + BiLSTM</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>66.29</td>
                  <td>86.97</td>
                  <td>25.08 ± 21.13</td>
                  <td>22.25 [24.29]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>16.02</td>
                  <td>17.38</td>
                  <td>14.93 ± 5.52</td>
                  <td>13.96 [8.17]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>1.11</td>
                  <td>1.19</td>
                  <td>15.46 ± 5.37</td>
                  <td>14.53 [6.77]</td>
                </tr>
                <tr>
                  <td rowspan="3">Proposed</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>43.55</td>
                  <td>49.81</td>
                  <td>16.04 ± 8.31</td>
                  <td>17.56 [12.48]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>7.12</td>
                  <td>9.13</td>
                  <td>6.51 ± 4.91</td>
                  <td>5.95 [7.16]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>0.53</td>
                  <td>0.64</td>
                  <td>7.35 ± 4.56</td>
                  <td>6.43 [6.77]</td>
                </tr>
              </tbody>
            </table>
			<table-wrap-foot>
            <fn>
              <p>LVET: Left ventricular ejection time; SV: stroke volume; CO: cardiac output; MAE: mean absolute error; RMSE: root mean square error.</p>
            </fn>
          </table-wrap-foot>
          </table-wrap>
          <p>The results show that, compared with the three learning-based baselines, including CNN-only, BiLSTM, and CNN + BiLSTM, the Proposed model achieves lower estimation errors for LVET, SV, and CO. Specifically, the Proposed model obtains mean MAPE values of 16.04%, 6.51%, and 7.35% for LVET, SV, and CO estimation, respectively. In comparison, the CNN-only model achieves 22.29%, 30.23%, and 30.67%, the BiLSTM model achieves 47.45%, 40.17%, and 40.54%, and the CNN + BiLSTM model achieves 25.08%, 14.93%, and 15.46%, respectively. These results indicate that relying solely on convolutional feature extraction, recurrent sequence modeling, or their combination remains insufficient for fully capturing the multimodal contextual information related to ejection events across ECG, ICG, and PCG signals. In contrast, the complete Proposed framework achieves lower parameter estimation errors under the current test set and unified input conditions. This result is consistent with the interpretation that multimodal temporal information and event-related context modeling may contribute to improved beat-level parameter estimation stability. However, since this comparison does not independently isolate the event-token mechanism, it should not be interpreted as direct evidence of the independent contribution of event-token.</p>
          <p>To further assess the independent contribution of local morphological information from the single-channel ICG signal to fiducial-point localization and hemodynamic parameter estimation, an additional ICG-only baseline model was included, with the results presented in <xref ref-type="table" rid="t13">Table 13</xref> and <xref ref-type="fig" rid="fig14">Figure 14</xref>. This model used only the single-channel ICG signal as input, without incorporating the electrical activity timing anchors provided by ECG or the valve-related mechanical event timing information provided by PCG. Apart from the input modality, the ICG-only baseline followed exactly the same data partitioning, preprocessing pipeline, number of training epochs, best-checkpoint selection strategy based on the validation set, and test-set evaluation procedure as those used in the main experiments, thereby ensuring a fair comparison across methods.</p>
          <fig id="fig14" position="float">
            <label>Figure 14</label>
            <caption>
              <p>Comparison of MAPE values for beat-level hemodynamic parameter estimation between the single-channel ICG-only model and the proposed model. (A-C) correspond to LVET, SV, and CO, respectively. The bar height represents the mean MAPE calculated from <italic>n</italic> = 61 valid test beats, and the error bars indicate mean ± SD.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.14.jpg" />
          </fig>
          <table-wrap id="t13">
            <label>Table 13</label>
            <caption>
              <p>Validation of the single-channel ICG model for direct feature-point localization and parameter estimation</p>
            </caption>
            <table frame="hsides" rules="groups">
              <thead>
                <tr>
                  <td style="border-bottom:1;">
                    <bold>Model</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Metric</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>n</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>MAE</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>RMSE</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>MAPE mean ± SD (%)</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>Median [IQR] (%)</bold>
                  </td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td rowspan="3">ICG-only</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>83.79</td>
                  <td>87.81</td>
                  <td>31.41 ± 7.80</td>
                  <td>34.85 [11.41]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>17.79</td>
                  <td>18.86</td>
                  <td>16.63 ± 5.18</td>
                  <td>16.29 [6.81]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>1.23</td>
                  <td>1.30</td>
                  <td>17.15 ± 5.23</td>
                  <td>17.60 [5.72]</td>
                </tr>
                <tr>
                  <td rowspan="3">Proposed</td>
                  <td>LVET</td>
                  <td>61</td>
                  <td>43.55</td>
                  <td>49.81</td>
                  <td>16.04 ± 8.31</td>
                  <td>17.56 [12.48]</td>
                </tr>
                <tr>
                  <td>SV</td>
                  <td>61</td>
                  <td>7.12</td>
                  <td>9.13</td>
                  <td>6.51 ± 4.91</td>
                  <td>5.95 [7.16]</td>
                </tr>
                <tr>
                  <td>CO</td>
                  <td>61</td>
                  <td>0.53</td>
                  <td>0.64</td>
                  <td>7.35 ± 4.56</td>
                  <td>6.43 [6.77]</td>
                </tr>
              </tbody>
            </table>
				<table-wrap-foot>
            <fn>
              <p>ICG: Impedance cardiography; LVET: Left ventricular ejection time; SV: stroke volume; CO: cardiac output; MAE: mean absolute error; RMSE: root mean square error.</p>
            </fn>
          </table-wrap-foot>
          </table-wrap>
          <p>The results showed that the ICG-only model achieved mean MAPEs of 31.41%, 16.63%, and 17.15% for LVET, SV, and CO, respectively, all of which were higher than the corresponding values of 16.04%, 6.51%, and 7.35% obtained by the proposed model. These findings suggest that reliance on local morphology from a single ICG channel alone is insufficient for stable identification of key events associated with AVO and closure, thereby limiting the accuracy of subsequent LVET, SV, and CO estimation. In contrast, the complete proposed multimodal framework achieves lower parameter estimation errors on the current test set, as shown in <xref ref-type="table" rid="t14">Table 14</xref>. This result is consistent with the interpretation that ECG-derived electrical timing anchors and PCG-derived valvular mechanical event information may provide complementary cues for ICG fiducial point localization.</p>
          <table-wrap id="t14">
            <label>Table 14</label>
            <caption>
              <p>Comparison of results between the best rule-based method and the DL + PCG model</p>
            </caption>
            <table frame="hsides" rules="groups">
              <thead>
                <tr>
                  <td style="border-bottom:1;">
                    <bold>Methods</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>LVET</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>SV</bold>
                  </td>
                  <td style="border-bottom:1;">
                    <bold>CO</bold>
                  </td>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>ECG + PCG-assisted</td>
                  <td>14.84</td>
                  <td>13.89</td>
                  <td>17.50</td>
                </tr>
                <tr>
                  <td>DL + PCG</td>
                  <td>16.04</td>
                  <td>6.51</td>
                  <td>7.35</td>
                </tr>
              </tbody>
            </table>
			<table-wrap-foot>
            <fn>
              <p>PCG: Phonocardiography; LVET: left ventricular ejection time; SV: stroke volume; CO: cardiac output.</p>
            </fn>
          </table-wrap-foot>
          </table-wrap>
        </sec>
        <sec id="sec3-3-3">
          <title>Impact of PCG on the deep learning model</title>
          <p>To further quantify the contribution of PCG within the deep learning framework, we compared two configurations: DL w/o PCG and DL + PCG. As shown in <xref ref-type="fig" rid="fig15">Figure 15</xref>, removing PCG leads to consistent performance degradation across all three metrics: the mean MAPE of LVET increases from 16.04% to 21.11%, SV from 6.51% to 8.31%, and CO from 7.35% to 9.00%. This result suggests that the valvular mechanical event-related temporal information contained in the PCG envelope can serve as complementary input for multimodal modeling and is consistent with the reduced parameter estimation errors observed under the current evaluation conditions, highlighting the potential value of PCG information for ICG-based parameter estimation tasks.</p>
          <fig id="fig15" position="float">
            <label>Figure 15</label>
            <caption>
              <p>Effect of PCG information on the beat-level parameter estimation performance of the deep learning model. (A and B) correspond to DL + PCG and DL w/o PCG, respectively. Each configuration is evaluated on <italic>n</italic> = 61 valid test beats. Each dot represents an individual beat; the width of the half violin indicates the kernel density distribution of MAPE values, the horizontal line represents the median, and the diamond indicates the mean.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.15.jpg" />
          </fig>
          <p>From a mechanistic perspective, the PCG envelope contains temporal information related to valvular mechanical events and can provide complementary mechanical references for ICG morphological features. Under the current data partitioning and model configuration, removing PCG results in increased estimation errors for LVET, SV, and CO. Correspondingly, the deep learning model integrating PCG achieves lower errors across all three metrics, with more pronounced improvements observed in SV and CO estimation. This ablation result indicates that PCG integration contributes to improved beat-level hemodynamic parameter estimation performance within the current complete model architecture and evaluation conditions. In the descriptive comparison of the four methods, the mean MAPE values for SV are 14.67%, 13.89%, 8.31%, and 6.51%, respectively, while those for CO are 18.35%, 17.50%, 9.00%, and 7.35%, respectively. Both metrics show an overall decreasing trend across the four configurations. LVET does not follow the same monotonic pattern, with mean MAPE values of 15.45%, 14.84%, 21.11%, and 16.04%, respectively. Nevertheless, within the conventional rule-based comparison and the deep learning comparison, the incorporation of PCG reduces the LVET error in both cases. Therefore, the current findings support the performance benefits of the complete multimodal framework and PCG integration under the conditions of this study, but they do not allow the observed improvements to be fully attributed to the event query mechanism or provide direct evidence that improved AVO/AVC localization is the primary cause of reduced downstream parameter estimation errors.</p>
          <p>To more clearly illustrate the performance gains introduced by multimodal fusion, <xref ref-type="fig" rid="fig16">Figure 16</xref> presents a heatmap of overall errors across all experimental settings. Compared with the darker regions corresponding to higher errors in conventional methods, the DL + PCG approach exhibits prominent bright regions - indicating lower errors - particularly for SV and CO. This visual contrast highlights the performance transition from rule-based matching to end-to-end feature learning.</p>
          <fig id="fig16" position="float">
            <label>Figure 16</label>
            <caption>
              <p>Heatmap of mean MAPE values for LVET, SV, and CO across the four experimental groups. Statistics for the conventional rule-based methods are based on <italic>n</italic> = 92 valid test beats, whereas statistics for the deep learning methods are based on <italic>n</italic> = 61 valid test beats. Color indicates the mean MAPE (%) of each method for the corresponding metric.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.16.jpg" />
          </fig>
          <p>Placing the four methods within a common coordinate system provides a more intuitive view of the combined contributions of PCG information and deep learning-based modeling. <xref ref-type="fig" rid="fig17">Figures 17</xref> and <xref ref-type="fig" rid="fig18">18</xref> present the results of the four methods from the perspectives of error distributions and metric-specific performance, respectively. Because the conventional rule-based and deep learning methods are evaluated on <italic>n</italic> = 92 and <InlineParagraph><italic>n</italic> = 61</InlineParagraph> valid test beats, respectively, their presentation within the same figures is intended primarily to describe the error characteristics of different configurations rather than to support paired statistical comparisons across method categories. For individual metrics, ECG + PCG-assisted shows numerically lower errors than ECG-only for LVET, SV, and CO. Within the same test-beat set used for the deep learning models, DL + PCG achieves lower mean MAPE values than DL w/o PCG for all three metrics, with more pronounced improvements in SV and CO. Overall, these results support the complementary value of PCG information and multimodal temporal modeling for beat-level parameter estimation under the current evaluation conditions.</p>
          <fig id="fig17" position="float">
            <label>Figure 17</label>
            <caption>
              <p>Raincloud plots of beat-level MAPE for LVET, SV, and CO across four methods. (A) Beat-level MAPE distribution for LVET. (B) Beat-level MAPE distribution for CO. (C) Beat-level MAPE distribution for SV. The conventional rule-based methods, ECG-only and ECG + PCG-assisted, are each evaluated on <italic>n</italic> = 92 valid test beats, whereas the deep learning methods, DL w/o PCG and DL + PCG, are each evaluated on <italic>n</italic> = 61 valid test beats. The scatter points represent the MAPE of individual valid beats, the density curves represent the error distributions, the boxplots indicate the median and interquartile range, and the diamonds denote the mean.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.17.jpg" />
          </fig>
          <fig id="fig18" position="float">
            <label>Figure 18</label>
            <caption>
              <p>Analysis of metric-wise MAPE distributions and mean error composition for different methods. (A) Violin plots show the beat-level MAPE distributions. (B) Stacked bar charts show the mean MAPE composition of LVET, SV, and CO for each method. Each colored segment represents the mean MAPE of one individual metric. The total value shown above each stacked bar is the sum of the mean MAPE values for LVET, SV, and CO for the corresponding method, and the black line connects these total values across methods.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.18.jpg" />
          </fig>
        </sec>
      </sec>
      <sec id="sec3-4">
        <title>Subject-independent generalization assessed by LOSO cross-validation</title>
        <p>To further address subject independence and model generalization, LOSO cross-validation was supplemented in this study. In each fold, one subject was completely held out as the independent test set, one subject was used as the validation set, and the remaining 15 subjects were used for model training, thereby avoiding cardiac beat segments from the same subject appearing simultaneously in the training and test sets. By iteratively treating each subject as the independent test subject, this setting provides a supplementary assessment of model generalization to unseen subjects. Unlike the fixed subject-level split used in the main experiment, LOSO cross-validation covers all 17 subjects, with each subject serving as the test subject once.</p>
        <p>In 17-fold LOSO validation, the DL + PCG model was evaluated for beat-level LVET, SV, and CO estimation under subject-independent testing in <xref ref-type="table" rid="t15">Table 15</xref>. When summarized at the subject level, the MAPE values for LVET, SV, and CO were 21.39 ± 13.87%, 27.42 ± 21.79%, and 26.69 ± 22.39%, respectively, with 95% confidence intervals of 14.26-28.52%, 16.22-38.63%, and and15.18-38.20%. These results indicate that the multimodal framework is feasible under strict subject-independent evaluation, while the observed inter-subject variability indicates that larger samples and more diverse populations are needed to improve and validate cross-subject generalization.</p>
        <table-wrap id="t15">
          <label>Table 15</label>
          <caption>
            <p>Subject-level performance of the DL + PCG model under LOSO cross-validation</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Metric</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Folds</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>MAPE mean ± SD (%)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>95%CI (%)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Median [IQR] (%)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>MAE mean ± SD</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>RMSE mean ± SD</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>LVET</td>
                <td>17</td>
                <td>21.39 ± 13.87</td>
                <td>14.26-28.52</td>
                <td>18.79<break />[14.99, 24.91]</td>
                <td>39.80 ± 21.14 ms</td>
                <td>45.50 ± 21.27 ms</td>
              </tr>
              <tr>
                <td>SV</td>
                <td>17</td>
                <td>27.42 ± 21.79</td>
                <td>16.22-38.63</td>
                <td>22.79<break />[14.39, 29.96]</td>
                <td>19.90 ± 11.25 mL</td>
                <td>21.44 ± 10.81 mL</td>
              </tr>
              <tr>
                <td>CO</td>
                <td>17</td>
                <td>26.69 ± 22.39</td>
                <td>15.18-38.20</td>
                <td>19.18<break />[12.53, 31.57]</td>
                <td>1.34 ± 0.79 L/min</td>
                <td>1.46 ± 0.78 L/min</td>
              </tr>
            </tbody>
          </table>
		  <table-wrap-foot>
            <fn>
              <p>MAPE: Mean absolute percentage error; LVET: Left ventricular ejection time; SV: stroke volume; CO: cardiac output; MAE: mean absolute error; RMSE: root mean square error.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec id="sec3-5">
        <title>Rationale for the fixed input window</title>
        <p>To justify the fixed input window, the temporal distributions of AVO, AVC, and conventional ICG B/X landmarks relative to the R peak were analyzed in <xref ref-type="table" rid="t16">Table 16</xref> and <xref ref-type="fig" rid="fig19">Figure 19</xref>. Across 129 HDF5 record files, 1,036 AVO and 1,069 AVC events were matched from HeartCycle, and the conventional morphology-based pipeline yielded 50 B points and 50 X points. AVC, B, and X all fell within the fixed window from 0.20 s before to 0.80 s after the R peak, with 100% coverage. AVO coverage was 99.52%, and its 99% distribution range was approximately 5.30-781.44 ms, still within the upper boundary of the window. Only five AVO events were outside the window, possibly due to boundary beats, event-matching errors, or rare extreme timing. These results support the use of a fixed 1.00 s R-peak-aligned window for beat-level modeling, while also indicating that atypical timing patterns may require further consideration in broader populations.</p>
        <fig id="fig19" position="float">
          <label>Figure 19</label>
          <caption>
            <p>Coverage of AVO, AVC, and B/X events within the fixed temporal window. The sample sizes are <italic>n</italic> = 1,036 for AVO, <italic>n</italic> = 1,069 for AVC, <italic>n</italic> = 50 for B points, and <italic>n</italic> = 50 for X points. The bars or annotated values indicate the proportion of events falling within the fixed time window from 0.20 s before to 0.80 s after the R peak.</p>
          </caption>
          <graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="jca6046.fig.19.jpg" />
        </fig>
        <table-wrap id="t16">
          <label>Table 16</label>
          <caption>
            <p>Temporal distribution and coverage of events relative to the R peak within the fixed input window</p>
          </caption>
          <table frame="hsides" rules="groups">
            <thead>
              <tr>
                <td style="border-bottom:1;">
                  <bold>Event</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Source</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>n</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Mean ± SD (ms)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Median [IQR] (ms)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>95% range (ms)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>99% range (ms)</bold>
                </td>
                <td style="border-bottom:1;">
                  <bold>Window coverage (missing n)</bold>
                </td>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>AVO</td>
                <td>HeartCycle</td>
                <td>1,036</td>
                <td>86.38 ± 98.81</td>
                <td>74.49 [51.11]</td>
                <td>15.07-192.54</td>
                <td>5.30-781.44</td>
                <td>99.52% (5)</td>
              </tr>
              <tr>
                <td>AVC</td>
                <td>HeartCycle</td>
                <td>1,069</td>
                <td>271.84 ± 50.47</td>
                <td>272.24 [66.97]</td>
                <td>177.42-370.11</td>
                <td>118.52-427.44</td>
                <td>100.00% (0)</td>
              </tr>
              <tr>
                <td>B</td>
                <td>Rule-based B</td>
                <td>50</td>
                <td>57.99 ± 20.76</td>
                <td>60.44 [40.27]</td>
                <td>30.22-90.62</td>
                <td>30.21-90.64</td>
                <td>100.00% (0)</td>
              </tr>
              <tr>
                <td>X</td>
                <td>Rule-based X</td>
                <td>50</td>
                <td>353.11 ± 3.94</td>
                <td>352.50 [4.65]</td>
                <td>347.44-361.77</td>
                <td>347.21-361.77</td>
                <td>100.00% (0)</td>
              </tr>
            </tbody>
          </table>
		  <table-wrap-foot>
            <fn>
              <p>AVO: Aortic valve opening; AVC: aortic valve closure.</p>
            </fn>
          </table-wrap-foot>
		 </table-wrap>
        <p>From the perspective of model generalization, the advantage of the fixed-window strategy is that it provides a unified input length and a clear R-peak-aligned reference, facilitating fusion of ECG, ICG, and PCG signals on the same time axis. At the same time, the 0.20 s window before the R peak preserves local baseline and early morphological information, whereas the 0.80 s window after the R peak covers major events such as AVO, AVC, the B point, and the X point. It should be noted that this strategy also has potential limitations. Under extreme heart rates, R-peak misdetection, event-label shifts, or severe signal drift, a small number of events may approach the window boundary or even fall outside the fixed window. In addition, because the fixed window is aligned only to the R peak and no time-scale normalization is performed according to the R-R interval, the model retains true event-delay differences across beats. However, this may also increase the difficulty of model learning under marked heart-rate variability. Therefore, the window setting adopted in this study is reasonable under the current HeartCycle data distribution, but future external validation should further assess its applicability in broader heart-rate ranges and pathological populations.</p>
      </sec>
    </sec>
    <sec id="sec4">
      <title>DISCUSSION</title>
      <p>This study establishes a research framework ranging from ICG waveform morphology analysis to multimodal feature fusion of ECG, ICG, and PCG to address the susceptibility of ICG fiducial point localization to waveform morphological variability in continuous noninvasive hemodynamic assessment. Clustering and classification analyses based on the HeartCycle dataset demonstrate pronounced morphological diversity, class overlap, and ambiguous boundaries among real-world ICG beats, indicating that relying solely on local morphological rules of ICG signals for B/X point localization remains challenging under complex waveform conditions. To address this issue, this study introduces ECG as an electrical timing anchor, utilizes the PCG envelope as complementary temporal information associated with valvular mechanical events, and integrates deep temporal modeling with an event query mechanism for beat-level LVET, SV, and CO estimation. Experimental results show that the complete multimodal framework achieves lower parameter estimation errors under the current evaluation conditions, with particularly improved stability in SV and CO estimation, indicating that multimodal temporal information fusion provides a feasible modeling strategy for alleviating morphological ambiguity in single-modality ICG analysis. It should be noted that the AVO, AVC, SV, and CO reference labels and values used in this study are obtained from publicly available HeartCycle HDF5 fields. Therefore, model evaluation remains affected by the reliability of the original event timestamps and hemodynamic reference values. Since the public dataset does not provide further information regarding beat-level annotation uncertainty, inter-annotator agreement, or repeated annotation variability, temporal resolution of event annotations, synchronization errors among devices, differences in event boundary determination, and resampling procedures may contribute to residual errors. Accordingly, the event fields provided by the public dataset are described as reference labels rather than error-free ground truth, and uncertainty in labels and reference values is considered an important factor affecting model performance interpretation.</p>
      <p>Further analysis shows that compared with the model without PCG, integrating PCG reduces SV and CO estimation errors, suggesting that the PCG envelope may provide complementary mechanical information associated with ejection event timing. This finding is consistent with the interpretation that multimodal temporal information, PCG-based mechanical event context, and event query-based localization design may jointly contribute to improved beat-level parameter estimation stability. Meanwhile, the current experiments mainly evaluate the overall performance of the complete multimodal framework under unified data partitioning and evaluation procedures. The independent contribution of the event-query mechanism and the error propagation relationship between AVO/AVC localization errors and SV/CO estimation errors require further quantification through more fine-grained component ablation and error association analyses. From a practical application perspective, the results of this study support the feasibility of the proposed method for beat-to-beat hemodynamic parameter estimation under the current evaluation conditions and provide a methodological foundation for future continuous noninvasive hemodynamic trend assessment.</p>
      <p>However, the findings of this study are primarily based on the HeartCycle public dataset and its current data distribution. Although the dataset contains multiple valid beats, and this study employs subject-level aggregated paired comparisons together with supplementary LOSO validation to strengthen the preliminary assessment of cross-subject applicability, the limited number of subjects and the single data source indicate that the current conclusions are most applicable to beat-level parameter estimation within this cohort. Future studies should further evaluate the external applicability of the model across different devices, acquisition conditions, patient populations, and multicenter cohorts. In addition, the reference values for AVO, AVC, SV, and CO are obtained from fields provided in the public dataset, and uncertainties in annotation accuracy and multimodal device synchronization may contribute to the observed errors. The available data also mainly support the evaluation of beat-level estimation performance. Incorporating controlled hemodynamic changes and long-term follow-up data in future work would facilitate further assessment of the model for continuous trend monitoring, within-subject change tracking, and analysis of the contributions of individual model components.</p>
     <sec id="sec4-1">
      <title>Conclusion</title>
      <p>This study establishes a comprehensive framework for ICG fiducial point localization and hemodynamic parameter estimation, covering waveform morphology analysis and multimodal deep fusion modeling. Clustering and classification analyses based on the HeartCycle dataset demonstrate pronounced morphological diversity, class overlap, and ambiguous boundaries among ICG waveforms under real beat conditions, suggesting that relying solely on local morphological rules from single-modality signals for B/X point localization remains limited under complex waveform conditions. Based on these findings, this study proposes an end-to-end deep learning model integrating ECG, ICG, and PCG and compares it with conventional rule-based methods and learning-based baseline models under a unified parameter derivation and evaluation framework. Experimental results demonstrate that the proposed multimodal framework can achieve beat-level LVET, SV, and CO estimation under the current evaluation conditions, with lower errors particularly observed in SV and CO estimation. The ablation experiments and baseline comparisons are consistent with the interpretation that PCG-based mechanical event context, multimodal temporal information, and event-related modeling may jointly contribute to improved parameter estimation stability, highlighting the potential of multimodal information fusion for ICG fiducial point localization and hemodynamic parameter estimation. Further LOSO validation demonstrates the feasibility of the framework for beat-level LVET, SV, and CO estimation under subject-independent testing conditions, providing experimental support for future studies involving larger cohorts, multicenter datasets, and subject-adaptive modeling. Overall, the current results primarily support the capability of the proposed method for beat-to-beat hemodynamic parameter estimation under the evaluated conditions and provide a methodological foundation for continuous noninvasive hemodynamic trend assessment. Reliable continuous trend monitoring and within-subject tracking of hemodynamic changes require further establishment through longitudinal data, dynamic physiological scenarios, and prospective studies. Future work will incorporate independent external datasets, multicenter validation, controlled hemodynamic state variation experiments, and fine-grained structural ablation analyses to systematically evaluate the model’s external generalization capability, within-subject change tracking ability, component contributions, and clinical applicability.</p>
    </sec>
	 </sec>
  </body>
  <back>
    <sec>
	 <title>DECLARATIONS</title>
      <sec>
        <title>Authors’ contributions</title>
        <p>Contributed to the study conception and design, data analysis and interpretation, and writing the manuscript: Ma S, Wang H, Cao F</p>
		<p>Performed data acquisition and provided administrative, technical, and material support: Xu Z, Yan Z, Zhao Z</p>
		<p>Contributed to manuscript revision and result interpretation during the revision: Wu N</p>
		<p>Commented on the research design, data analysis, and supervision of the study: Wang H, Cao F</p>
		<p>All authors reviewed the manuscript and approved the submitted version.</p>
      </sec>
      <sec>
        <title>Availability of data and materials</title>
        <p>The data used in this study were obtained from the publicly available HeartCycle dataset, HeartCycle: A comprehensive dataset of synchronized ICG and echocardiography for accurate hemodynamic predictions (version 1.0.0), hosted on PhysioNet (<uri xlink:href="https://physionet.org/content/heartcycle/1.0.0/">https://physionet.org/content/heartcycle/1.0.0/</uri>). The dataset can be accessed through PhysioNet under the Open Data Commons Attribution License v1.0. No new clinical dataset was generated in this study. The HeartCycle dataset contains synchronized ICG, ECG, echocardiography, photoplethysmography, and heart sound recordings, together with related hemodynamic and physiological parameters. The processed data and analysis scripts generated during the current study are available from the corresponding author upon reasonable request.</p>
      </sec>
      <sec>
        <title>AI and AI-assisted tools statement</title>
        <p>Not applicable.</p>
      </sec>
      <sec>
        <title>Financial support and sponsorship</title>
        <p>None.</p>
      </sec>
      <sec>
        <title>Conflicts of interest</title>
        <p>Cao F is an Associate Editor of <italic>The Journal of Cardiovascular Aging</italic>. Cao F is also the Guest Editor of the special issue entitled “Artificial Intelligence in Cardiovascular Aging and Disease” in <italic>The Journal of Cardiovascular Aging</italic>. Cao F was not involved in any steps of editorial processing, notably including reviewers’ selection, manuscript handling, and decision making, while the other authors have declared that they have no conflicts of interest.</p>
      </sec>
	   <sec>
        <title>Ethical approval and consent to participate</title>
        <p>This study involved a secondary analysis of the publicly available and de-identified HeartCycle v1.0.0 dataset. The original HeartCycle study was approved by the Ethics Committee of the University of Coimbra Hospital (reference CES-238) and was conducted in accordance with the Declaration of Helsinki. Written informed consent was obtained from all participants before data collection. No new participants were recruited, and no identifiable personal information was accessed in the present study.</p>
      </sec>
      <sec>
        <title>Consent for publication</title>
        <p>Not applicable.</p>
      </sec>
	   <sec>
        <title>Copyright</title>
        <p>© The Author(s) 2026.</p>
      </sec>
	  
    </sec>
	<ref-list>
      <ref id="B1">
        <label>1</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kubicek</surname>
              <given-names>WG</given-names>
            </name>
            <name>
              <surname>Patterson</surname>
              <given-names>RP</given-names>
            </name>
            <name>
              <surname>Witsoe</surname>
              <given-names>DA</given-names>
            </name>
          </person-group>
          <article-title>Impedance cardiography as a noninvasive method of monitoring cardiac function and other parameters of the cardiovascular system</article-title>
          <source>Ann N Y Acad Sci</source>
          <year>2006</year>
          <volume>170</volume>
          <fpage>724</fpage>
          <lpage>32</lpage>
          <pub-id pub-id-type="doi">10.1111/j.1749-6632.1970.tb17735.x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B2">
        <label>2</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Lababidi</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Ehmke</surname>
              <given-names>DA</given-names>
            </name>
            <name>
              <surname>Durnin</surname>
              <given-names>RE</given-names>
            </name>
            <name>
              <surname>Leaverton</surname>
              <given-names>PE</given-names>
            </name>
            <name>
              <surname>Lauer</surname>
              <given-names>RM</given-names>
            </name>
          </person-group>
          <article-title>The first derivative thoracic impedance cardiogram</article-title>
          <source>Circulation</source>
          <year>1970</year>
          <volume>41</volume>
          <fpage>651</fpage>
          <lpage>8</lpage>
          <pub-id pub-id-type="doi">10.1161/01.cir.41.4.651</pub-id>
          <pub-id pub-id-type="pmid">5437409</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B3">
        <label>3</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Sherwood(chair)</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Allen</surname>
              <given-names>MT</given-names>
            </name>
            <name>
              <surname>Fahrenberg</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Kelsey</surname>
              <given-names>RM</given-names>
            </name>
            <name>
              <surname>Lovallo</surname>
              <given-names>WR</given-names>
            </name>
            <name>
              <surname>Van Doornen</surname>
              <given-names>LJ</given-names>
            </name>
          </person-group>
          <article-title>Methodological guidelines for impedance cardiography</article-title>
          <source>Psychophysiology</source>
          <year>2007</year>
          <volume>27</volume>
          <fpage>1</fpage>
          <lpage>23</lpage>
          <pub-id pub-id-type="doi">10.1111/j.1469-8986.1990.tb02171.x</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B4">
        <label>4</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Scholte</surname>
              <given-names>NTB</given-names>
            </name>
            <name>
              <surname>Van Ravensberg</surname>
              <given-names>AE</given-names>
            </name>
            <name>
              <surname>Shakoor</surname>
              <given-names>A</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>A scoping review on advancements in noninvasive wearable technology for heart failure management</article-title>
          <source>NPJ Digit Med</source>
          <year>2024</year>
          <volume>7</volume>
          <fpage>279</fpage>
          <pub-id pub-id-type="doi">10.1038/s41746-024-01268-5</pub-id>
          <pub-id pub-id-type="pmid">39396094</pub-id>
          <pub-id pub-id-type="pmcid">PMC11470936</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B5">
        <label>5</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Petek</surname>
              <given-names>BJ</given-names>
            </name>
            <name>
              <surname>Al-Alusi</surname>
              <given-names>MA</given-names>
            </name>
            <name>
              <surname>Moulson</surname>
              <given-names>N</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Consumer wearable health and fitness technology in cardiovascular medicine: JACC state-of-the-art review</article-title>
          <source>J Am Coll Cardiol</source>
          <year>2023</year>
          <volume>82</volume>
          <fpage>245</fpage>
          <lpage>64</lpage>
          <pub-id pub-id-type="doi">10.1016/j.jacc.2023.04.054</pub-id>
          <pub-id pub-id-type="pmid">37438010</pub-id>
          <pub-id pub-id-type="pmcid">PMC10662962</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B6">
        <label>6</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Jamieson</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Chico</surname>
              <given-names>TJA</given-names>
            </name>
            <name>
              <surname>Jones</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Chaturvedi</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Hughes</surname>
              <given-names>AD</given-names>
            </name>
            <name>
              <surname>Orini</surname>
              <given-names>M</given-names>
            </name>
          </person-group>
          <article-title>A guide to consumer-grade wearables in cardiovascular clinical care and population health for non-experts</article-title>
          <source>NPJ Cardiovasc Health</source>
          <year>2025</year>
          <volume>2</volume>
          <fpage>44</fpage>
          <pub-id pub-id-type="doi">10.1038/s44325-025-00082-6</pub-id>
          <pub-id pub-id-type="pmid">40909206</pub-id>
          <pub-id pub-id-type="pmcid">PMC12404996</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B7">
        <label>7</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Bernstein</surname>
              <given-names>DP</given-names>
            </name>
          </person-group>
          <article-title>Impedance cardiography: pulsatile blood flow and the biophysical and electrodynamic basis for the stroke volume equations</article-title>
          <source>J Electr Bioimpedance</source>
          <year>2009</year>
          <volume>1</volume>
          <fpage>2</fpage>
          <lpage>17</lpage>
          <pub-id pub-id-type="doi">10.5617/jeb.51</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B8">
        <label>8</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Árbol</surname>
              <given-names>JR</given-names>
            </name>
            <name>
              <surname>Perakakis</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Garrido</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Mata</surname>
              <given-names>JL</given-names>
            </name>
            <name>
              <surname>Fernández-Santaella</surname>
              <given-names>MC</given-names>
            </name>
            <name>
              <surname>Vila</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>Mathematical detection of aortic valve opening (B point) in impedance cardiography: A comparison of three popular algorithms</article-title>
          <source>Psychophysiology</source>
          <year>2016</year>
          <volume>54</volume>
          <fpage>350</fpage>
          <lpage>7</lpage>
          <pub-id pub-id-type="doi">10.1111/psyp.12799</pub-id>
          <pub-id pub-id-type="pmid">27914174</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B9">
        <label>9</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Trybek</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Sobotnicka</surname>
              <given-names>E</given-names>
            </name>
            <name>
              <surname>Wawrzkiewicz-Jałowiecka</surname>
              <given-names>A</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>A new method of identifying characteristic points in the impedance cardiography signal based on empirical mode decomposition</article-title>
          <source>Sensors</source>
          <year>2023</year>
          <volume>23</volume>
          <fpage>675</fpage>
          <pub-id pub-id-type="doi">10.3390/s23020675</pub-id>
          <pub-id pub-id-type="pmid">36679466</pub-id>
          <pub-id pub-id-type="pmcid">PMC9861967</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B10">
        <label>10</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Pan</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Tompkins</surname>
              <given-names>WJ</given-names>
            </name>
          </person-group>
          <article-title>A real-time QRS detection algorithm</article-title>
          <source>IEEE Trans Biomed Eng</source>
          <year>1985</year>
          <volume>BME-32</volume>
          <fpage>230</fpage>
          <lpage>6</lpage>
          <pub-id pub-id-type="doi">10.1109/tbme.1985.325532</pub-id>
          <pub-id pub-id-type="pmid">3997178</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B11">
        <label>11</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Fuller</surname>
              <given-names>HD</given-names>
            </name>
          </person-group>
          <article-title>Evaluation of left ventricular function by impedance cardiography: a review</article-title>
          <source>Prog Cardiovasc Dis</source>
          <year>1994</year>
          <volume>36</volume>
          <fpage>267</fpage>
          <lpage>73</lpage>
          <pub-id pub-id-type="doi">10.1016/s0033-0620(05)80035-0</pub-id>
          <pub-id pub-id-type="pmid">8284433</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B12">
        <label>12</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Karpiel</surname>
              <given-names>I</given-names>
            </name>
            <name>
              <surname>Richter-Laskowska</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Feige</surname>
              <given-names>D</given-names>
            </name>
            <name>
              <surname>Gacek</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Sobotnicki</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>An effective method of detecting characteristic points of impedance cardiogram verified in the clinical pilot study</article-title>
          <source>Sensors</source>
          <year>2022</year>
          <volume>22</volume>
          <fpage>9872</fpage>
          <pub-id pub-id-type="doi">10.3390/s22249872</pub-id>
          <pub-id pub-id-type="pmid">36560238</pub-id>
          <pub-id pub-id-type="pmcid">PMC9782651</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B13">
        <label>13</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Xie</surname>
              <given-names>Y</given-names>
            </name>
            <name>
              <surname>Yu</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Xie</surname>
              <given-names>Q</given-names>
            </name>
          </person-group>
          <article-title>Motion impedance cardiography denoising method based on canonical correlation analysis and coherence analysis</article-title>
          <source>Biomed Signal Process Control</source>
          <year>2023</year>
          <volume>86</volume>
          <fpage>105300</fpage>
          <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105300</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B14">
        <label>14</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Li</surname>
              <given-names>X</given-names>
            </name>
            <name>
              <surname>Ni</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Ji</surname>
              <given-names>Z</given-names>
            </name>
          </person-group>
          <article-title>ICG signal denoising based on ICEEMDAN and PSO-VMD methods</article-title>
          <source>Phys Eng Sci Med</source>
          <year>2024</year>
          <volume>47</volume>
          <fpage>1547</fpage>
          <lpage>56</lpage>
          <pub-id pub-id-type="doi">10.1007/s13246-024-01467-0</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B15">
        <label>15</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Colominas</surname>
              <given-names>MA</given-names>
            </name>
            <name>
              <surname>Schlotthauer</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Torres</surname>
              <given-names>ME</given-names>
            </name>
          </person-group>
          <article-title>Improved complete ensemble EMD: a suitable tool for biomedical signal processing</article-title>
          <source>Biomed Signal Process Control</source>
          <year>2014</year>
          <volume>14</volume>
          <fpage>19</fpage>
          <lpage>29</lpage>
          <pub-id pub-id-type="doi">10.1016/j.bspc.2014.06.009</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B16">
        <label>16</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Dragomiretskiy</surname>
              <given-names>K</given-names>
            </name>
            <name>
              <surname>Zosso</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <article-title>Variational mode decomposition</article-title>
          <source>IEEE Trans Signal Process</source>
          <year>2014</year>
          <volume>62</volume>
          <fpage>531</fpage>
          <lpage>44</lpage>
          <pub-id pub-id-type="doi">10.1109/tsp.2013.2288675</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B17">
        <label>17</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Kumari</surname>
              <given-names>PD</given-names>
            </name>
            <name>
              <surname>Singh</surname>
              <given-names>KM</given-names>
            </name>
            <name>
              <surname>Mayaluri</surname>
              <given-names>ZL</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>A hybrid variational mode decomposition framework for enhanced cardiac output estimation using impedance cardiography</article-title>
          <source>Sci Rep</source>
          <year>2025</year>
          <volume>15</volume>
          <fpage>25784</fpage>
          <pub-id pub-id-type="doi">10.1038/s41598-025-09948-2</pub-id>
          <pub-id pub-id-type="pmid">40670468</pub-id>
          <pub-id pub-id-type="pmcid">PMC12267840</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B18">
        <label>18</label>
        <nlm-citation publication-type="confproc">
          <person-group person-group-type="author">
            <name>
              <surname>Pale</surname>
              <given-names>U</given-names>
            </name>
            <name>
              <surname>Muller</surname>
              <given-names>N</given-names>
            </name>
            <name>
              <surname>Arza</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Atienza</surname>
              <given-names>D</given-names>
            </name>
          </person-group>
          <comment>ReBeatICG: real-time low-complexity beat-to-beat impedance cardiogram delineation algorithm. In 2021 43rd Annual International Conference of the IEEE Engineering in Medicine &amp; Biology Society (EMBC); 2021 Nov 1-5; Mexico. IEEE; 2021. pp. 5618-24.</comment>
          <pub-id pub-id-type="doi">10.1109/embc46164.2021.9630170</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B19">
        <label>19</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Inan</surname>
              <given-names>OT</given-names>
            </name>
            <name>
              <surname>Migeotte</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Park</surname>
              <given-names>K</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Ballistocardiography and seismocardiography: a review of recent advances</article-title>
          <source>IEEE J Biomed Health Inform</source>
          <year>2015</year>
          <volume>19</volume>
          <fpage>1414</fpage>
          <lpage>27</lpage>
          <pub-id pub-id-type="doi">10.1109/jbhi.2014.2361732</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B20">
        <label>20</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Bhattacharya</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Santucci</surname>
              <given-names>F</given-names>
            </name>
            <name>
              <surname>Jankovic</surname>
              <given-names>M</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Cardiac time intervals under motion using bimodal chest E-tattoos and multistage processing</article-title>
          <source>IEEE Trans Biomed Eng</source>
          <year>2025</year>
          <volume>72</volume>
          <fpage>413</fpage>
          <lpage>24</lpage>
          <pub-id pub-id-type="doi">10.1109/tbme.2024.3454067</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B21">
        <label>21</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Lin</surname>
              <given-names>DJ</given-names>
            </name>
            <name>
              <surname>Kimball</surname>
              <given-names>JP</given-names>
            </name>
            <name>
              <surname>Zia</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Ganti</surname>
              <given-names>VG</given-names>
            </name>
            <name>
              <surname>Inan</surname>
              <given-names>OT</given-names>
            </name>
          </person-group>
          <article-title>Reducing the impact of external vibrations on fiducial point detection in seismocardiogram signals</article-title>
          <source>IEEE Trans Biomed Eng</source>
          <year>2022</year>
          <volume>69</volume>
          <fpage>176</fpage>
          <lpage>85</lpage>
          <pub-id pub-id-type="doi">10.1109/tbme.2021.3090376</pub-id>
          <pub-id pub-id-type="pmid">34161234</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B22">
        <label>22</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Shandhi</surname>
              <given-names>MMH</given-names>
            </name>
            <name>
              <surname>Fan</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Heller</surname>
              <given-names>JA</given-names>
            </name>
            <name>
              <surname>Etemadi</surname>
              <given-names>M</given-names>
            </name>
            <name>
              <surname>Klein</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Inan</surname>
              <given-names>OT</given-names>
            </name>
          </person-group>
          <article-title>Estimation of changes in intracardiac hemodynamics using wearable seismocardiography and machine learning in patients with heart failure: a feasibility study</article-title>
          <source>IEEE Trans Biomed Eng</source>
          <year>2022</year>
          <volume>69</volume>
          <fpage>2443</fpage>
          <lpage>55</lpage>
          <pub-id pub-id-type="doi">10.1109/tbme.2022.3147066</pub-id>
          <pub-id pub-id-type="pmid">35100106</pub-id>
          <pub-id pub-id-type="pmcid">PMC9347221</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B23">
        <label>23</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Zhou</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Huang</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>H</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>Camera seismocardiogram based monitoring of left ventricular ejection time</article-title>
          <source>IEEE Trans Biomed Eng</source>
          <year>2025</year>
          <volume>72</volume>
          <fpage>2609</fpage>
          <lpage>22</lpage>
          <pub-id pub-id-type="doi">10.1109/tbme.2025.3548090</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B24">
        <label>24</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Jiménez-González</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <article-title>Timing the opening and closure of the aortic valve using a phonocardiogram envelope: a performance test for systolic time intervals measurement</article-title>
          <source>Physiol Meas</source>
          <year>2021</year>
          <volume>42</volume>
          <fpage>025004</fpage>
          <pub-id pub-id-type="doi">10.1088/1361-6579/abe0fe</pub-id>
          <pub-id pub-id-type="pmid">33705357</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B25">
        <label>25</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Zang</surname>
              <given-names>J</given-names>
            </name>
            <name>
              <surname>An</surname>
              <given-names>Q</given-names>
            </name>
            <name>
              <surname>Li</surname>
              <given-names>B</given-names>
            </name>
            <name>
              <surname>Zhang</surname>
              <given-names>Z</given-names>
            </name>
            <name>
              <surname>Gao</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Xue</surname>
              <given-names>C</given-names>
            </name>
          </person-group>
          <article-title>A novel wearable device integrating ECG and PCG for cardiac health monitoring</article-title>
          <source>Microsyst Nanoeng</source>
          <year>2025</year>
          <volume>11</volume>
          <fpage>7</fpage>
          <pub-id pub-id-type="doi">10.1038/s41378-024-00858-3</pub-id>
          <pub-id pub-id-type="pmid">39814701</pub-id>
          <pub-id pub-id-type="pmcid">PMC11735617</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B26">
        <label>26</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Illueca</surname>
              <given-names>Fernandez E</given-names>
            </name>
            <name>
              <surname>Couceiro</surname>
              <given-names>R</given-names>
            </name>
            <name>
              <surname>Abtahi</surname>
              <given-names>F</given-names>
            </name>
            <etal />
          </person-group>
          <article-title>HeartCycle: a comprehensive dataset of synchronized impedance cardiography and echocardiography for accurate hemodynamic predictions (version 1.0.0)</article-title>
          <source>PhysioNet</source>
          <year>2025</year>
          <pub-id pub-id-type="doi">10.13026/z865-eb23</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B27">
        <label>27</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Savitzky</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Golay</surname>
              <given-names>MJE</given-names>
            </name>
          </person-group>
          <article-title>Smoothing and differentiation of data by simplified least squares procedures</article-title>
          <source>Anal Chem</source>
          <year>2002</year>
          <volume>36</volume>
          <fpage>1627</fpage>
          <lpage>39</lpage>
          <pub-id pub-id-type="doi">10.1021/ac60214a047</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B28">
        <label>28</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Mondal</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Bhattacharya</surname>
              <given-names>P</given-names>
            </name>
            <name>
              <surname>Saha</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <article-title>An automated tool for localization of heart sound components S1, S2, S3 and S4 in pulmonary sounds using Hilbert transform and Heron’s formula</article-title>
          <source>SpringerPlus</source>
          <year>2013</year>
          <volume>2</volume>
          <fpage>512</fpage>
          <pub-id pub-id-type="doi">10.1186/2193-1801-2-512</pub-id>
          <pub-id pub-id-type="pmid">24255827</pub-id>
          <pub-id pub-id-type="pmcid">PMC3825056</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B29">
        <label>29</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Thalmayer</surname>
              <given-names>A</given-names>
            </name>
            <name>
              <surname>Zeising</surname>
              <given-names>S</given-names>
            </name>
            <name>
              <surname>Fischer</surname>
              <given-names>G</given-names>
            </name>
            <name>
              <surname>Kirchner</surname>
              <given-names>J</given-names>
            </name>
          </person-group>
          <article-title>A robust and real-time capable envelope-based algorithm for heart sound classification: validation under different physiological conditions</article-title>
          <source>Sensors</source>
          <year>2020</year>
          <volume>20</volume>
          <fpage>972</fpage>
          <pub-id pub-id-type="doi">10.3390/s20040972</pub-id>
          <pub-id pub-id-type="pmid">32054136</pub-id>
          <pub-id pub-id-type="pmcid">PMC7070375</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B30">
        <label>30</label>
        <nlm-citation publication-type="journal">
          <person-group person-group-type="author">
            <name>
              <surname>Sakoe</surname>
              <given-names>H</given-names>
            </name>
            <name>
              <surname>Chiba</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <article-title>Dynamic programming algorithm optimization for spoken word recognition</article-title>
          <source>IEEE Trans Acoust Speech Signal Process</source>
          <year>1978</year>
          <volume>26</volume>
          <fpage>43</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1109/tassp.1978.1163055</pub-id>
        </nlm-citation>
      </ref>
      <ref id="B31">
        <label>31</label>
        <nlm-citation publication-type="book">
          <person-group person-group-type="author">
            <name>
              <surname>Kaufman</surname>
              <given-names>L</given-names>
            </name>
            <name>
              <surname>Rousseeuw</surname>
              <given-names>PJ</given-names>
            </name>
          </person-group>
          <comment>Finding groups in data: an introduction to cluster analysis. Hoboken: John Wiley &amp; Sons; 1990.</comment>
          <pub-id pub-id-type="doi">10.1002/9780470316801</pub-id>
        </nlm-citation>
      </ref>
    </ref-list>
  
  </back>
</article>