@inproceedings{6309c709bbbc438ebbfcbb34f9d846d4,
title = "Generating and mixing feature sets from language models for sentiment classification",
abstract = "This paper presents methods for mixing feature sets in sentence-level sentiment analysis where a sentence is classified into one of three classes: positive, negative, and neutral. Motivated by the need to classify sentences in Korean whose sentiment-revealing expressions tend to have different effects according to their syntactic categories, we employed a language modeling (LM) approach with 162 different LMs based on syntactic categories that are effectively combined with a Logistic Regression classifier. The experimental results show that this approach significantly outperforms clue-based SVM classifiers. The enumeration of feature types arising from the LMs for the Logistic Regression classifier allowed us to show that domain specific models can be smoothed with a general model and that attaching a syntactic category to a feature helps improving effectiveness. The classification results are further improved by applying a clue-based classifier. The rationale behind this two-step process is to classify sentences with a relatively conservative classifier in picking positive and negative sentences and to apply a high-precision classifier to the sentences in the neutral class.",
keywords = "Polarity classification, Sentiment analysis, Text categorization",
author = "Yoonjae Jeong and Youngho Kim and Seongchan Kim and Myaeng, \{Sung Hyon\} and Oh, \{Hyo Jung\}",
year = "2009",
doi = "10.1109/NLPKE.2009.5313746",
language = "English",
isbn = "9781424445387",
series = "2009 International Conference on Natural Language Processing and Knowledge Engineering, NLP-KE 2009",
booktitle = "2009 International Conference on Natural Language Processing and Knowledge Engineering, NLP-KE 2009",
note = "2009 International Conference on Natural Language Processing and Knowledge Engineering, NLP-KE 2009 ; Conference date: 24-09-2009 Through 27-09-2009",
}