@article{shahabikargar2025language,
  title = {Language Model-Enhanced Feature Engineering Framework for Customer Churn Analysis},
  author = {Maryam SHAHABIKARGAR and Amin BEHESHTI and Saleh AFZOON and Jin FOO and Xuyun ZHANG and Nasrin SHABANI},
  year = 2025,
  url = {https://ibimapublishing.com/p-articles/45AI/2025/4526625/},
  journal = {Communications of International Proceedings},
  volume = 2025 (20),
  doi = doi.org/10.5171/2025.4526625,
  abstract = {Customer churn remains a major concern across industries, especially with the rising importance of customer retention over acquisition. While prior research has focused heavily on structured data, the potential of customer-generated textual data, such as chat logs and feedback, remains underutilized. This study addresses this gap by introducing a novel feature engineering framework that leverages Language Models (LMs), to extract meaningful insights from unstructured text data for churn prediction. We propose a multi-stage pipeline that combines domain expertise with LM capabilities to generate interaction features, sentiment labels, emotional tone scores, and topic- based features from customer chat data. Additionally, we introduce a new composite metric, the Normalized-weighted Churn Score, which integrates expert-assigned topic weights with language model outputs. The framework was evaluated using a churn dataset containing structured and unstructured data. Results show that incorporating LM-enhanced features significantly boosts model performance across multiple classifiers. Notably, models using our enriched feature outperformed traditional baselines, achieving an F1-score increase of over 26%. The findings emphasize the critical role of text analytics and hybrid feature engineering in advancing churn pre- diction and offer a scalable approach for integrating domain knowledge with modern NLP techniques.},
  keywords = {Customer Churn Analysis, Feature Engineering, Language Models, Textual Data Processing},
  note = Article ID: 4526625
}
