@inproceedings{fd95f63f690c4a55aff08b9cb2836602,
title = "RGB-T Multi-modal Visual Question Answering in Nighttime and Adverse Environment",
abstract = "Visual Question Answering (VQA) models that rely only on RGB inputs often fail in nighttime and adverse environments due to poor illumination and semantic loss. To address this, we propose an RGB-T VQA framework that integrates visible and thermal infrared (TIR) modalities. The framework contains two key modules: a Cross-Modal Guided Attention (CGA) that uses thermal cues to refine RGB features, and a Thermal-Semantic Prior (TSP) that compensates for the limited semantics of TIR data. In addition, we construct a large-scale RGB-T VQA dataset covering diverse nighttime, low-light, and adverse weather scenes.",
keywords = "RGB-T fusion, Visual question answering (VQA), low-light vision, multimodal learning, thermal infrared (TIR)",
author = "Songyuan Yang and Fan Yang and Biwen Yang and Jing Zhao and Yongqiang Sun and Ruiheng Zhang",
note = "Publisher Copyright: {\textcopyright} 2025 IEEE.; 4th International Conference on Advanced Sensing and Intelligent Manufacturing, ASIM 2025 ; Conference date: 31-10-2025 Through 02-11-2025",
year = "2025",
doi = "10.1109/ASIM67379.2025.11512835",
language = "English",
series = "Proceeding of the 2025 4th International Conference on Advanced Sensing and Intelligent Manufacturing, ASIM 2025",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
booktitle = "Proceeding of the 2025 4th International Conference on Advanced Sensing and Intelligent Manufacturing, ASIM 2025",
address = "United States",
}