
@Article{cmc.2026.084782,
AUTHOR = {Zishi Li, Xiaodong Huang},
TITLE = {DCHF: Dual-Stream Cooperative Perception with Hierarchical Fusion Network for Micro-Expression Recognition},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {89},
YEAR = {2026},
NUMBER = {2},
PAGES = {--},
URL = {http://www.techscience.com/cmc/v89n2/68774},
ISSN = {1546-2226},
ABSTRACT = {Micro-expression recognition (MER) is a challenging task because micro-expressions are extremely short in duration, weak in intensity, and often distributed over subtle local facial regions. Existing methods either rely on handcrafted descriptors with limited representation capacity or focus on single-stream deep models that do not fully exploit structural facial dependency and cross-modal complementarity. As a result, they often fail to effectively capture subtle local muscle activations, model structured facial interactions, and robustly integrate complementary motion and appearance cues, especially under the weak-motion conditions characteristic of micro-expressions. To address these limitations, this paper proposes a Dual-Stream Cooperative Perception with Hierarchical Fusion Network (DCHF) for MER. The proposed framework jointly models temporal optical flow and spatial apex frames through a dual-stream architecture. The framework contains three key components. Adaptive Centroid-Aware Feature Extraction enhances key muscle activation regions through adaptive centroid-aware local attention while preserving global semantic context. Vertical Ipsilateral Dependency explicitly models same-side upper–lower facial coupling. The Hierarchical Dual-Stream Fusion Module progressively fuses motion and appearance information from local to global semantic levels. Extensive experiments on the Spontaneous Micro-expression Corpus (SMIC), Chinese Academy of Sciences Micro-Expression II (CASME II), Spontaneous Actions and Micro-Movements (SAMM), and their composite setting demonstrate that DCHF achieves competitive performance. In particular, on the composite benchmark, DCHF obtains 0.9121 Unweighted F1-score (UF1) and 0.9182 Unweighted Average Recall (UAR), improving the strongest compared UF1 and UAR results by 0.70 and 2.49 percentage points, respectively. These improvements suggest that adaptive centroid-aware local modeling, structure-aware ipsilateral dependency learning, and hierarchical dual-stream fusion are complementary and effective for robust MER under cross-dataset variation.},
DOI = {10.32604/cmc.2026.084782}
}



