@article{Tzachristas2026, 
author = {Ioannis Tzachristas and Santhanakrishnan Narayanan and Constantinos Antoniou},
title = {LLM-PDM: An LLM Persona-Driven Method for replicating personal mobility preferences at scale},
year = {2026},
journal = {Communications in Transportation Research},
volume = {6},
number = {1},
pages = {9640004},
keywords = {synthetic data generation, travel survey, large language models (LLMs), Persona-Driven Modeling (PDM), prompt engineering, synthetic populations, mobility data},
url = {https://www.sciopen.com/article/10.26599/COMMTR.2026.9640004},
doi = {10.26599/COMMTR.2026.9640004},
abstract = {Traditional travel surveys are costly, time-consuming and face declining response rates, motivating the exploration of artificial data generation methods. In this research, we propose a novel Persona-Driven Method (PDM) for generating synthetic mobility survey data via large language models (LLMs). The method defines representative personas—each characterized by specific sociodemographic attributes—and prompts an LLM to emulate survey respondents with these personas. A guided prompting strategy is introduced to calibrate the synthetic data distributions so that they closely match real-world population statistics. We evaluate the approach on the German MiD 2017 (Mobilität in the Deutschland 2017) dataset. The quality of the LLM-PDM-generated synthetic data is assessed against ground truth data via a comprehensive set of metrics, including the mean absolute error (MAE), root mean square error (RMSE), Jensen‒Shannon distance (JSD), entropy, conditional entropy and the Earth Mover’s distance (EMD). The empirical results demonstrate that the LLM-PDM approach produces high-fidelity synthetic populations that preserve key distributions and relationships present in real data. Across the case studies, the LLM-PDM method achieves low distributional errors (e.g., MAE &lt; 3%) and captures important joint patterns, significantly outperforming a number of LLM baselines.}
}