@article{Cui2026, 
author = {Junqi Cui and Weijia Li and Enoch Chi Ngai Lim and Xiaoqin Wu and Fang Chen and Chi Eung Danforn Lim},
title = {From TCMBank to translational hypotheses: A machine-learning framework for prioritizing Chinese medicinal herbs for liver diseases},
year = {2026},
journal = {Journal of Traditional Chinese Medical Sciences},
volume = {13},
number = {3},
pages = {310-318},
keywords = {Liver disease, Traditional Chinese medicine, Machine learning, Herb prioritization},
url = {https://www.sciopen.com/article/10.1016/j.jtcms.2026.06.003},
doi = {10.1016/j.jtcms.2026.06.003},
abstract = {ObjectiveTo develop a machine-learning framework integrating TCMBank-derived liver disease seed curation with structured herb-level annotations to prioritize herbs with potential relevance to liver disease.MethodsLiver-focused non-tumor disease seeds were curated from TCMBank to identify 356 annotation-supported positive herbs. Herb-level features were derived from structured TCMBank annotations. Six unweighted machine-learning models were trained using the original 356-positive/8835-background positive-unlabeled dataset. Performance was assessed using the area under the receiver operating characteristic curve and the area under the precision-recall curve (PR-AUC), with PR-AUC interpreted relative to the baseline prevalence of 0.039. Candidate herbs were further refined through consensus prioritization, cross-model concordance, and translational evidence evaluation.ResultsA total of 124 curated liver disease seeds and 9191 TCMBank herb records were retained. The final modeling dataset comprised 356 annotation-defined positive herbs and 8835 unlabeled background herbs, corresponding to a positive prevalence of 0.039. Model performance was evaluated using the 356-positive positive-unlabeled dataset. PR-AUC baselines, candidate rankings, and validation-priority scores were generated within and aligned with the final modeling framework.ConclusionsThis study establishes a reproducible machine-learning framework for prioritizing candidate herbs in liver disease research and provides a data-driven strategy for translating large-scale database resources for Chinese medicine into experimentally-testable hypotheses. The prioritized candidates should be regarded as computational hypotheses requiring staged pharmacological, hepatobiliary, and safety validation rather than as evidence of established clinical efficacy.}
}