@article{ZHANG2026, 
author = {Lu ZHANG and Jing LU and Xiao-Qing GU},
title = {How to Conduct Large-Scale Assessment of Creative Thinking?——Exploring a “Human-in-the-Loop” Human-Machine Collaborative Scoring Mode Supported by Large Language Model},
year = {2026},
journal = {Modern Educational Technology},
volume = {36},
number = {4},
pages = {83-91},
keywords = {creative thinking, large language model, “human-in-the-loop”, human-machine collaborative scoring},
url = {https://www.sciopen.com/article/10.3969/j.issn.1009-8097.2026.04.009},
doi = {10.3969/j.issn.1009-8097.2026.04.009},
abstract = {Creative thinking, as a key element in the cultivation of top-notch innovative talents, its scientific assessment is an important direction for the reform of educational evaluation. The PISA 2022 Creative Thinking Assessment Framework provides a mature reference for large-scale standardized evaluation. However, manual scoring has problems such as difficult standard unification, low efficiency and poor quality guarantee, which urgently needs technological empowerment. Based on this, the paper constructed a “human-in-the-loop” human-machine collaborative scoring mode supported by large language model, and explained the implementation mechanisms of applying this mode for creative thinking assessment from three aspects of theoretical basis, key technologies and practical paths. Subsequently, taking the Shanghai student creative thinking assessment project as a case study, this paper verified the effectiveness of this mode through comparative analysis of relevant data from both manual and human-machine collaborative scoring methods. It was found that both scoring methods were reliable, while the human-machine collaborative scoring showed higher factor loading values and discrimination values compared to manual scoring, and reduced the workload per human rate by approximately 66.7% in the human-machine collaborative scoring. The adaptability of human-computer collaborative assessment was influenced by the degree of task structuring, and expression-type questions required deeper human intervention. The research in this paper provided a human-machine collaborative pathway and empirical basis that can balance quality and efficiency for large-scale standardized assessment of creative thinking, and can also offer a reference for the standardized promotion of more extensive and higher-order thinking assessment enabled by artificial intelligence.}
}