@article{Cheng2026, 
author = {Qimin Cheng and Jiajun Ling and Yingjie Du and Qunshan Zhao},
title = {Toward comprehensive traffic scene understanding: a benchmark and detector for traffic object detection in smart city surveillance},
year = {2026},
journal = {Geo-Spatial Information Science},
volume = {29},
number = {4},
pages = {2369-2392},
keywords = {Traffic object detection, intelligent transportation systems, surveillance scenarios, smart city},
url = {https://www.sciopen.com/article/10.1080/10095020.2025.2542964},
doi = {10.1080/10095020.2025.2542964},
abstract = {Traffic object detection serves as a critical enabler of intelligent transportation systems (ITS). However, it faces multi-dimensional challenges in complex scenarios, including object heterogeneity, scene adaptation, and the trade-off between accuracy and real-time performance. To mitigate these issues, we propose TSO-DETR, a transformer-based detection network for traffic surveillance. It integrates four key modules: a local pattern learning module that enhances the representation of small or structured objects through directional edge-aware learning; a class-aware masker module for improving category-level feature discrimination; a illumination correction module for adapting to varying lighting conditions; and an aspect-ratio-aware loss that refines localization for elongated objects. To mitigate the scarcity of standard benchmark with more comprehensive elements, we construct CCTRIB-DET, featuring 150k annotated instances across 9 categories. It includes both dynamic and static elements, covers diverse conditions, and offers rich surveillance viewpoints, making it a standardized and versatile dataset for real-world evaluation. TSO-DETR is evaluated across multiple datasets. On both the constructed CCTRIB-DET dataset (achieving 74.42% AP) and the public SEU_PML benchmark (31.70% AP), TSO-DETR demonstrates comparable performance to the state-of-the-art CO-DETR, while delivering a 5× acceleration in inference speed. It also performs competitively on the vehicle-mounted detection BDD100K dataset, low-light detection ExDark, and traffic sign dataset TT100K, demonstrating its effectiveness, robustness, and cross-domain generalization.}
}