@article{Cao2026, 
author = {Jingwen Cao and Hairong Lv and Jianye Xue and Siyuan Wang and Haijie Wang and Ming Zhao and Weixiao Wang},
title = {Knowledge representation and storage for large-scale intelligent models: Paradigms, system architectures, and governance toward verifiable knowledge systems},
year = {2026},
journal = {Cybernetics and Intelligence},
keywords = {Knowledge representation, Knowledge storage, Large language models, Foundation models, Knowledge graphs, Retrieval-augmented generation, Neuro-symbolic AI, Provenance, Verifiable knowledge systems},
url = {https://www.sciopen.com/article/10.26599/CAI.2026.9390020},
doi = {10.26599/CAI.2026.9390020},
abstract = {Knowledge representation and storage have long constituted the epistemic infrastructure of artificial intelligence. In the era of large-scale intelligent models, however, the problem has been fundamentally reconfigured. Knowledge is no longer only a matter of symbolic encoding or database persistence; it has become a system-level issue involving parametric memory, retrieval-augmented access, provenance, authority, and behavioral control. Existing studies have made substantial progress on specific subtopics such as knowledge graphs, retrieval-augmented generation, tool use, and model editing, yet a unified treatment of representation, storage, and governance remains lacking. This survey revisits the field from that broader systems perspective. We clarify the conceptual boundaries among data, information, knowledge, evidence, memory, and intelligence, and distinguish representation from storage as two coupled but non-identical design problems. We then review the evolution of knowledge paradigms, spanning symbolic and logic-based systems, knowledge graphs and structured knowledge bases, database-centric infrastructures, vectorized retrieval systems, parametric knowledge in large models, and recent neuro-symbolic and tool-augmented hybrids. Building on this review, we identify a recurrent systemic failure in current retrieval-dominated architectures—functional collapse—where facts, rules, evidence, and authority are flattened into homogeneous retrievable context. We argue that the central challenge is therefore not merely how to represent more knowledge, but how to organize knowledge as a governed system asset that can support factual grounding, justification, and authority regulation. On this basis, we formulate a governance-oriented view of verifiable knowledge systems and outline a research agenda toward auditable, updateable, and trustworthy knowledge infrastructures for next-generation intelligent systems.}
}