@article{Guo2025, 
author = {Zhen Guo and Ying Zhou and Jun Ye and Yongxu Hou},
title = {Desensitization of Private Text Dataset Based on Gradient Strategy Trans-WTGAN},
year = {2025},
journal = {Tsinghua Science and Technology},
volume = {30},
number = {5},
pages = {2081-2096},
keywords = {desensitization, gradient strategy, Transformer and Wasserstein Text convolutional Generative Adversarial Network (Trans-WTGAN), usability},
url = {https://www.sciopen.com/article/10.26599/TST.2024.9010155},
doi = {10.26599/TST.2024.9010155},
abstract = {Privacy-sensitive data encounter immense security and usability challenges in processing, analyzing, and sharing. Meanwhile, traditional privacy data desensitization methods suffer from issues such as poor quality and low usability after desensitization. Therefore, a text data desensitization model that combines Transformer and Wasserstein Text convolutional Generative Adversarial Network (Trans-WTGAN) is proposed. Transformer as the generator and its self-attention mechanism can handle long-range dependencies, enabling the generated of higher-quality text; Text Convolutional Neural Network (TextCNN) integrates the idea of Wasserstein as the discriminator to enhance the stability of model training; and the strategy gradient scheme of reinforcement learning is employed. Reinforcement learning utilizes the policy gradient scheme as the updating method of generator parameters, ensuring the generated data retains the original key features and maintains a certain level of usability. The experimental results indicate that the proposed model scheme holds a greater advantage over existing methods in terms of text quality and structural consistency, can guarantee the desensitization effect, and ensures the usability of the privacy-sensitive data to a certain extent after desensitization, facilitates the simulation of the development environment for the use of real data and the analysis and sharing of data.}
}