@article{Zhang2026, 
author = {Jiarui Zhang and Jierui Chen and Shiyu Fang and Chao Huang and Peng Hang and Jian Sun},
title = {Toward the intelligence evaluation of high-level autonomous vehicles: Subjective-objective mapping method via LLM},
year = {2026},
journal = {Communications in Transportation Research},
volume = {6},
number = {3},
pages = {9640041},
keywords = {autonomous vehicles (AVs), intelligence evaluation, large language models (LLMs), interaction behavior, scenario-based testing},
url = {https://www.sciopen.com/article/10.26599/COMMTR.2026.9640041},
doi = {10.26599/COMMTR.2026.9640041},
abstract = {With the advent of the commercialization phase for autonomous vehicles (AVs), the evaluation of their intelligence has become essential for regulators. However, existing evaluation methods are still largely traditional and experience-based. Therefore, this study proposed a subjective-objective mapping evaluation (SOME) method to evaluate the intelligence of high-level autonomous driving systems (ADSs). First, a five-dimensional evaluation metric system was developed to represent the overall performance of AVs during testing or actual driving. Next, the performance of AVs in a real-world driving dataset was evaluated based on the large language model (LLM). This approach could significantly enhance evaluation efficiency, achieving a passing rate of 86.75% in the Turing test. Finally, a deep neural network with an attention mechanism was trained using quantified metrics and an LLM-based evaluation label to serve as the evaluation model. Ablation experiments were conducted on both the LLM-based evaluation strategy and the evaluation model, demonstrating the necessity of each module. During the application phase, the evaluation model's effectiveness and reliability in scenario-based testing were validated using data from the OnSite Autonomous Driving Challenge results, and comparative analysis was conducted against traditional evaluation methods. The results show that our model's evaluations align closely with those of human experts and outperform traditional evaluation methods.}
}