@inproceedings{4793e6e9050d473aa5ef98571f1c03eb,
title = "A multi-information dual-layer cross-attention model for esophageal fistula prognosis",
abstract = "Esophageal fistula (EF) is a critical and life-threatening complication following radiotherapy treatment for esophageal cancer (EC). Albeit tabular clinical data contains other clinically valuable information, it is inherently different from CT images and the heterogeneity among them may impede the effective fusion of multi-modal data and thus degrade the performance of deep learning methods. However, current methodologies do not explicitly address this limitation. To tackle this gap, we present an adaptive multi-information dual-layer cross-attention (MDC) model using both CT images and tabular clinical data for early-stage EF detection before radiotherapy. Our MDC model comprises a clinical data encoder, an adaptive 3D Trans-CNN image encoder, and a dual-layer cross-attention (DualCrossAtt) module. The Image Encoder utilizes both CNN and transformer to extract multi-level local and global features, followed by global depth-wise convolution to remove the redundancy from these features for robust adaptive fusion. To mitigate the heterogeneity among multi-modal features and enhance fusion effectiveness, our DualCrossAtt applies the first layer of a cross-attention mechanism to perform alignment between the features of clinical data and images, generating commonly attended features to the second-layer cross-attention that models the global relationship among multi-modal features for prediction. Furthermore, we introduce a contrastive learning-enhanced hybrid loss function to further boost performance. Comparative evaluations against eight state-of-the-art multi-modality predictive models demonstrate the superiority of our method in EF prediction, with potential to assist personalized stratification and precision EC treatment planning.",
keywords = "Attention Networks, Multi-modal Data Fusion, Predictive Model",
author = "Jianqiao Zhang and Hao Xiong and Qiangguo Jin and Tian Feng and Jiquan Ma and Ping Xuan and Peng Cheng and Zhiyuan Ning and Zhiyu Ning and Changyang Li and Linlin Wang and Hui Cui",
year = "2024",
doi = "10.1007/978-3-031-72086-4\_3",
language = "English",
isbn = "9783031720857",
series = "Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)",
publisher = "Springer, Springer Nature",
pages = "25--35",
editor = "Linguraru, \{Marius George\} and Qi Dou and Aasa Feragen and Stamatia Giannarou and Ben Glocker and Karim Lekadir and Schnabel, \{Julia A.\}",
booktitle = "Medical Image Computing and Computer Assisted Intervention – MICCAI 2024",
address = "United States",
note = "International Conference on Medical Image Computing and Computer-Assisted Intervention (27th : 2024) , MICCAI 2024 ; Conference date: 06-10-2024 Through 10-10-2024",
}