@article{Yan2026, 
author = {Lawrence K. Q. Yan and Qian Niu and Ming Li and Yichao Zhang and Caitlyn Heqi Yin and Cheng Fei and Benji Peng and Ziqian Bi and Pohsun Feng and Keyu Chen and Tianyang Wang and Yunze Wang and Silin Chen and Ming Liu and Junyu Liu and Xinyuan Song and Riyang Bao and Zekun Jiang and Ziyuan Qin},
title = {Large Language Model Benchmarks in Medical Tasks},
year = {2026},
journal = {Medicine Advances},
volume = {4},
number = {3},
pages = {316-341},
keywords = {benchmark datasets, clinical summarization, electronic health records, large language models, medical artificial intelligence, medical imaging, multimodal data},
url = {https://www.sciopen.com/article/10.1002/med4.70085},
doi = {10.1002/med4.70085},
abstract = {With the increasing application of large language models (LLMs) in the medical domain, evaluating these models' performance using benchmark datasets has become crucial. This paper presents a comprehensive survey of various benchmark datasets used in medical LLM tasks. These datasets span multiple modalities including text, image, and multimodal benchmarks, focusing on various aspects of medical knowledge such as electronic health records, doctor–patient dialogues, medical question answering, and medical image captioning. The survey categorizes the datasets by modality and examines their significance, data structure, and roles in model development and evaluation across tasks such as diagnostic support, report generation, and predictive decision support. Representative resources include Medical Information Mart for Intensive Care Ⅲ (MIMIC‐Ⅲ), MIMIC‐Ⅳ, BioASQ, PubMedQA, and CheXpert, which provide data and evaluation settings for research in clinical NLP, medical question answering, and chest‐radiograph interpretation. This paper summarizes the challenges and opportunities in leveraging these benchmarks for advancing multimodal medical intelligence, emphasizing the need for datasets with a greater degree of language diversity, structured omics data, and innovative approaches to synthesis. This synthesis is intended to inform future research on the applications of LLMs in medicine and medical artificial intelligence.}
}