@article{Feng2026, 
author = {Weidan Feng and Shuaishuai Tai and Fangdong Liu and Yating Nie and Jiaoping Zhang and Guangyu Liu and Wubin Wang and Zhen Yue and Yan Li and Shouping Yang and Junyi Gai and Xiaodong Fang and Jianbo He},
title = {An efficient machine-learning framework for genomic selection of optimal crosses in soybean germplasm population},
year = {2026},
journal = {The Crop Journal},
volume = {14},
number = {4},
pages = {1388-1398},
keywords = {Soybean breeding, Optimal cross design, Genomic selection, Machine learning, Transgressive segregation prediction},
url = {https://www.sciopen.com/article/10.1016/j.cj.2026.03.003},
doi = {10.1016/j.cj.2026.03.003},
abstract = {Genomic selection (GS) has provided a comprehensive framework for efficient breeding by linking phenotypes to genome-wide markers. However, research on GS has predominantly focused on improving genotype-to-phenotype prediction models, often overlooking optimal cross design, which determines the potential of progeny selection and plays a critical role in crop breeding. In this study, an efficient GS framework, EMLGP (ensemble machine-learning for genomic prediction), was proposed for optimal cross design in crop breeding. EMLGP first employs machine-learning algorithms to train precise genotype-to-phenotype prediction models in a germplasm population and then integrates with genome simulations to predict optimal crosses in a breeding population. GS model training of 14 soybean traits demonstrated that EMLGP achieved superior performance, with the highest prediction accuracy (correlation coefficient) reaching 0.92. The prediction accuracy showed a maximum improvement of 35.85% over the classical GBLUP method. Further simulation studies confirmed that EMLGP exhibited robust performance under conditions of small-to-moderate sample sizes (300–5000), low-to-moderate trait heritabilities (0.4–0.6), and complex genetic architectures (100 causal loci). Validation using real data of rice, maize, cotton, sorghum, and switchgrass consistently affirmed EMLGP’s superiority, outperforming GBLUP and deep learning methods. Among the 14 soybean traits analyzed, 13 traits exhibited transgressive segregation potential in the progeny. Specifically, seed linolenic acid content in the northern China showed the highest recombination potential, exceeding the maximum parental value by 16.89%. In conclusion, EMLGP optimizes parental selection and phenotypic prediction, offering a robust framework for efficient, intelligence-driven crop breeding.}
}