@article{Li2026, 
author = {Hong-Bin Li and Yin-Hong Tian and Gui-Wen Wang and Zhi-Bin He and Lin-Bo Shao and Jin Lai},
title = {Enhancing reservoir characterization via lithology-constrained machine learning methods: The deep Jurassic Sangonghe tight sandstones in the Turpan-Hami Basin, China},
year = {2026},
journal = {Petroleum Science},
volume = {23},
number = {8},
pages = {4505-4527},
keywords = {Deep tight sandstone reservoir, Parameter prediction, Machine learning, SHAP-based interpretability, Sangonghe Formation, Turpan-Hami Basin},
url = {https://www.sciopen.com/article/10.1016/j.petsci.2026.05.021},
doi = {10.1016/j.petsci.2026.05.021},
abstract = {Reliable porosity prediction in deep tight sandstones remains a persistent challenge due to complex lithological heterogeneity and weak correlations between conventional logs and reservoir parameters. This study systematically evaluates the performance of four predictive models, namely SVM, CatBoost, 2-D CNN and LSTM, in predicting NMR-derived effective porosity using nine conventional log curves and lithology as inputs. A total of 190 test samples from well J7-2-3H and blind well validation from the Jurassic Sangonghe Formation in the Turpan-Hami Basin were employed for performance benchmarking. The lithologies are mainly siltstone, fine-grained sandstone, medium-grained sandstone, coarse-grained sandstone, and sandy conglomerate. Results show that CatBoost consistently outperforms both traditional method and the other two deep learning models, achieving the highest accuracy (R2 = 0.94, MAE = 0.24%) and demonstrating superior robustness. SHAP analysis was used to interpret model behavior and confirmed that CatBoost provides physically consistent predictions, assigning geologically reasonable weights to key features such as gamma ray, density, and lithology. The model–data compatibility and the structured treatment of discrete geological information are essential for enhancing predictive performance. Furthermore, lithology serves not only as an auxiliary input but also as a soft petrophysical constraint that enables improved model generalization across heterogeneous reservoirs. Despite the greater architectural complexity of deep learning models, their limited capacity to handle categorical features under data-scarce conditions results in suboptimal performance compared to CatBoost. This study presents a comparative modeling framework and interpretability strategy for evaluating reservoir parameter prediction models, emphasizing the value of incorporating geological priors in data-driven workflows. Medium- and coarse-grained sandstones exhibit relatively high porosity, with the latter showing a slightly superior porosity range. The proposed approach offers practical guidance for algorithm selection and feature engineering in intelligent reservoir characterization. This methodology enables precise calculation of reservoir parameters for deep tight sandstone reservoirs in the absence of NMR logging tools, thereby providing both theoretical frameworks and technical support for reservoir evaluation and reserve assessment of deep tight sandstones.}
}