
@Article{cmc.2026.084514,
AUTHOR = {Zarnab Kausar, Shaheryar Najam, Hadeel Alsolai, Bayan Alabdullah, Fatimah Alhayan, Ahmad Jalal, Hui Liu},
TITLE = {Deep Hand Segmentation and Multi-Modal Gesture Recognition for Human-Robot Interaction via 3D Volumetric Encoding},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/28235},
ISSN = {1546-2226},
ABSTRACT = {Hand gesture recognition (HGR) is essential for Human–Robot Interaction (HRI) but remains challenging due to variations in hand shape, motion, viewpoint, illumination, and background, while vision-based methods often suffer from sensitivity to skin tone, occlusions, deformations, and limited interpretability. To address these issues, we propose a unified framework integrating deep learning, geometry-driven analysis, and temporal motion modeling. We introduce Z-HandSegNet framework, involving a U-Net with a ResNet-34 encoder for robust hand segmentation, and the Ellipse-Guided Geometric Finger Segmentation and Keypoint Extraction (EG-FSKE) method, which decomposes hand silhouettes into palm and finger regions using distance transforms, Gaussian Mixture Models, ellipse fitting, and a fragment recovery strategy for occluded and deformed fingers. To capture the hierarchical structure and dense motion, adaptive octree-based volumetric representation and Ef-RAFT optical flow are used for dynamic modeling. These representations are combined with saliency and ridge-based descriptors into a multimodal feature and classified with a Transformer encoder for modelling temporal dependencies. The evaluations on three datasets (Jester, IPN Hand and EgoGesture) achieve state-of-the-art performance with accuracies of 93.22%, 92.34%, and 94.17%, respectively, and precision, recall and F1-scores above 0.87. Ablation studies confirm the contribution of each component, particularly Z-HandSegNet and EG-FSKE, highlighting the robustness, interpretability, and potential of the proposed approach for future real-time HRI applications.},
DOI = {10.32604/cmc.2026.084514}
}



