@article{Hu2026, 
author = {Rongsheng Hu and Yuan Liu and Runwei Guan and Yicheng Di and Jiayu Bao and Yi Jin and Hongjian Shi and Yuan Hong and Lin Yu and Ruhui Ma and Rajkumar Buyya},
title = {VQA-G Annotator: A Self-Improving Agentic Framework for High-Fidelity Grounded Dataset Synthesis},
year = {2026},
journal = {Tsinghua Science and Technology},
keywords = {Visual Question Answering, Visual Grounding, Synthetic Data Generation, Agentic Framework, Spatial Reasoning},
url = {https://www.sciopen.com/article/10.26599/TST.2026.9010076},
doi = {10.26599/TST.2026.9010076},
abstract = {As artificial intelligence systems increasingly rely on multi-source perception and cross-modal learning for human-like visual understanding, their outputs should be both semantically valid and spatially grounded. However, large-scale vision-language benchmarks with precise grounding remain costly to construct, while existing synthetic pipelines often suffer from distribution drift, error propagation, and weak spatial verification. We introduce VQA-G Annotator, a self-improving agentic framework that combines distribution-aware planning, multi-agent orchestration, and spatial reasoning verification in a closed-loop generation, evaluation, and refinement process. Experiments across four benchmark sources show that VQA-G Annotator achieves an average VQAScore of 0.88 for semantic alignment and an Acc@0.5 of 0.57 for spatial grounding, comparing favorably with strong automatic baselines. Downstream and human evaluations further support the utility of the synthesized data 18 for scalable and trustworthy grounded visual understanding.}
}