@article{Ning2026, 
author = {Xin Ning and Kai Zhao and Dan Li and Jin Ning and Weijun Li and Baoli Lu},
title = {Pixel-Level Registration Method and Dataset Construction for Multimodal Image Data in Intelligent Cockpits},
year = {2026},
journal = {Tsinghua Science and Technology},
keywords = {intelligent cockpit, dataset, multimodal perception, pixel-level registration, object detection},
url = {https://www.sciopen.com/article/10.26599/TST.2026.9010048},
doi = {10.26599/TST.2026.9010048},
abstract = {Driver behavior monitoring is fundamental to safety assurance and interaction optimization in intelligent cockpits. To address key challenges in existing research, including limited modality diversity, cross-view spatial mis-alignment, and insufficient robustness of multimodal perception, we construct a multimodal intelligent-cockpit image dataset consisting of synchronously captured RGB, infrared (IR), and depth images. We also propose a two-stage registration method for dataset preprocessing that combines geometric mapping and mask-guided inpainting to achieve pixel-level alignment across modalities. Using the constructed dataset, we evaluate the impact of different modality combinations on object detection. Results on two baseline multimodal detectors show that the tri-modal setting significantly outperforms RGB-only input in both accuracy and robustness, particularly under challenging lighting conditions, with mAP@0.5:0.95 gains of 36.8% and 36.3%, respectively. These findings verify the effectiveness of the proposed dataset for multimodal object detection and driver behavior understanding. The proposed dataset provides a valuable benchmark for multimodal perception in intelligent cockpits and facilitates future research on driver monitoring and active safety.}
}