
@Article{cmc.2026.086985,
AUTHOR = {Bayan Alabdullah, Muhammad Waqas Ahmed, Mohammad Shorfuzzaman, Jasem Almotiri, Mohammed Alonazi, Ahmad Jalal},
TITLE = {Attention-Guided Cross-Modal Transformer for Multimodal SAR-Optical Image Fusion and Flood Change Detection},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/27796},
ISSN = {1546-2226},
ABSTRACT = {Multimodal data fusion and deep learning have opened new frontiers in the analysis of complex visual data acquired from heterogeneous sensing systems. Flood inundation mapping represents one of the most demanding applications in this domain, requiring robust interpretation of complementary but conflicting image modalities under severe real-world constraints. This paper presents CAG-Transformer, a novel multimodal AI architecture for bi-temporal flood change detection through intelligent fusion of Sentinel-1 SAR and Sentinel-2 multispectral imagery. Three tightly integrated contributions address the core challenges of heterogeneous multimodal image analysis. A Change Attention Gate (CAG) performs adaptive channel-wise representation learning, selectively amplifying flood-relevant spectral and backscatter variations while suppressing temporally static scene content. A Cross-Modal Transformer (CMT) bottleneck employs multi-head self-attention to model long-range spatial dependencies and enable context-aware reasoning across heterogeneous sensor representations capabilities fundamentally beyond convolutional fusion. A differentiable soft Dice loss ensures stable gradient flow under the severe class imbalance inherent to real-world flood datasets. Evaluated on the Ombria multimodal benchmark, CAG-Transformer achieves IoU = 0.7774, Dice = 0.8894, and AUC-ROC = 0.9768, outperforming single-modality and conventional fusion baselines. Cross-event validation on Albania (IoU = 0.7647) and Timor (IoU = 0.7313) confirms generalization across environmentally and spatially distinct flood. With 3.5 million parameters and sub-100 ms inference on freely available Copernicus data, the framework offers an efficient and deployable solution for near-real-time flood monitoring and multimodal image-based environmental assessment.},
DOI = {10.32604/cmc.2026.086985}
}



