@article{Dou2025, 
author = {Cheng-Feng Dou and Ying Zhang and Zhi Jin and Wen-Pin Jiao and Hai-Yan Zhao and Yong-Qiang Zhao and Zheng-Wei Tao},
title = {Exploring LLM-Based Data Synthesis Strategies for Aligning Medical Consultation Preferences},
year = {2025},
journal = {Journal of Computer Science and Technology},
volume = {40},
number = {6},
pages = {1485-1498},
keywords = {large language model (LLM), medical dialogue, reinforcement learning from artificial intelligence feedback (RLAIF), standardized patient testing},
url = {https://www.sciopen.com/article/10.1007/s11390-025-4929-7},
doi = {10.1007/s11390-025-4929-7},
abstract = {This research explores the application of reinforcement learning from artificial intelligence feedback (RLAIF) techniques to enhance healthcare consultation models, with the aim of addressing the challenges associated with preference-aligned data synthesis while reducing the dependence on medical experts. Specifically, we investigate the use of RLAIF in the generation of medical dialogues, focusing on two primary challenges: accurately reflecting doctors’ preferences and the unreliability of existing automated assessment systems. To address these issues, we propose a two-stage approach for synthesizing preference-aligned datasets. In the first stage, we leverage the dialogue continuation capabilities of a large language model to sample diverse, contextually aligned dialogue branches, employing one-shot learning for intervention. The second stage involves modeling doctors’ preferences through both outcome and process feedback. For outcome feedback, a rule-based reward system is utilized, whereas a planning-based reward strategy is employed for process feedback. To validate our approach, we develop the Chinese Standardized Patient Test (CSPT) dataset that emphasizes user guiding, instruction following, and synthesis ability, and construct an objective assessment system based on standardized patient testing. Experimental results demonstrate that our data synthesis approach performs well across five datasets, achieving a 17.6% improvement in diagnostic accuracy with outcome feedback and a 23.3% improvement with process feedback.}
}