文本情感分析
·
一、实训要求
1、掌握文本情感分析的基本概念,专业术语;
2、完成基于情感词典的情感分析。
3、完成基于文本分类的情感分析。
4、完成基于SnowNLP库的情感分析。
二、实训环境
- 装有Anaconda 的本地计算机,在Jupyter Notebook 或 PyCharm上完成相关实验任务。
- 安装jieba、SnowNLP、NLTK库。
三、过程截图
1、完成基于情感词典的情感分析。
import re
import jieba
import codecs
from collections import defaultdict # 导入collections用于创建空白词典
def seg_word(sentence):
seg_list = jieba.cut(sentence)
seg_result = []
for word in seg_list:
seg_result.append(word)
stopwords = set()
stopword = codecs.open('../data/stopwords.txt', 'r',
encoding='utf-8') # 加载停用词
for word in stopword:
stopwords.add(word.strip())
stopword.close()
return list(filter(lambda x: x not in stopwords, seg_result))
def sort_word(word_dict):
sen_file = open('../data/BosonNLP_sentiment_score.txt', 'r+',
encoding='utf-8') # 加载Boson情感词典
sen_list = sen_file.readlines()
sen_dict = defaultdict() # 创建词典
for s in sen_list:
s = re.sub('\n', '', s) # 去除每行最后的换行符
if s:
sen_dict[s.split(' ')[0]] = s.split(' ')[1]
not_file = open('../data/否定词.txt', 'r+',
encoding='utf-8') # 加载否定词词典
not_list = not_file.readlines()
for i in range(len(not_list)):
not_list[i] = re.sub('\n', '', not_list[i])
degree_file = open('../data/程度副词(中文).txt', 'r+',
encoding='utf-8') # 加载程度副词词典
degree_list = degree_file.readlines()
degree_dic = defaultdict()
for d in degree_list:
d = re.sub('\n', '', d)
if d:
degree_dic[d.split(' ')[0]] = d.split(' ')[1]
sen_file.close()
degree_file.close()
not_file.close()
sen_word = dict()
not_word = dict()
degree_word = dict()
for word in word_dict.keys():
if word in sen_dict.keys() and word not in not_list and word not in degree_dic.keys():
sen_word[word_dict[word]] = sen_dict[word] # 情感词典中的包含分词结果的词
elif word in not_list and word not in degree_dic.keys():
not_word[word_dict[word]] = -1 # 程度副词词典中的包含分词结果的词
elif word in degree_dic.keys():
degree_word[word_dict[word]] = degree_dic[word]
return sen_word, not_word, degree_word # 返回分类结果
def list_to_dict(word_list):
data = {}
for x in range(0, len(word_list)):
data[word_list[x]] = x
return data
def socre_sentiment(sen_word, not_word, degree_word, seg_result):
W = 1 # 初始化权重
score = 0
sentiment_index = -1 # 情感词下标初始化
for i in range(0, len(seg_result)):
if i in sen_word.keys():
score += W * float(sen_word[i])
sentiment_index += 1 # 下一个情感词
for j in range(len(seg_result)):
if j in not_word.keys():
score *= -1 # 否定词反转情感
elif j in degree_word.keys():
score *= float(degree_word[j]) # 乘以程度副词
return score
def setiment(sentence):
seg_list = seg_word(sentence)
sen_word, not_word, degree_word = sort_word(list_to_dict(seg_list))
score = socre_sentiment(sen_word, not_word, degree_word, seg_list)
return seg_list, sen_word, not_word, degree_word, score
if __name__ == '__main__':
print(setiment('我今天特别开心'))
print(setiment('我今天很开心、非常兴奋'))
print(setiment('我昨天开心,今天不开心'))

2、完成基于文本分类的情感分析。
import nltk.classify as cf
import nltk.classify.util as cu
import jieba
def setiment(sentences):
pos_data = []
with open('../data/pos.txt', 'r+', encoding='utf-8') as pos: # 读取积极评论
while True:
words = pos.readline()
if words:
positive = {} # 创建积极评论的词典
words = jieba.cut(words) # 对评论数据结巴分词
for word in words:
positive[word] = True
pos_data.append((positive, 'POSITIVE')) # 对积极词赋予POSITIVE标签
else:
break
neg_data = []
with open('../data/neg.txt', 'r+', encoding='utf-8') as neg: # 读取消极评论
while True:
words = neg.readline()
if words:
negative = {} # 创建消极评论的词典
words = jieba.cut(words) # 对评论数据结巴分词
for word in words:
negative[word] = True
neg_data.append((negative, 'NEGATIVE')) # 对消极词赋予NEGATIVE标签
else:
break
pos_num, neg_num = int(len(pos_data) * 0.8), int(len(neg_data) * 0.8)
train_data = pos_data[: pos_num] + neg_data[: neg_num] # 抽取80%数据
test_data = pos_data[pos_num: ] + neg_data[neg_num: ] # 剩余20%数据
model = cf.NaiveBayesClassifier.train(train_data)
ac = cu.accuracy(model, test_data)
print('准确率为:' + str(ac))
tops = model.most_informative_features() # 信息量较大的特征
print('\n信息量较大的前10个特征为:')
for top in tops[: 10]:
print(top[0])
for sentence in sentences:
feature = {}
words = jieba.cut(sentence)
for word in words:
feature[word] = True
pcls = model.prob_classify(feature)
sent = pcls.max() # 情绪面标签(POSITIVE或NEGATIVE)
prob = pcls.prob(sent) # 情绪程度
print('\n','‘',sentence,'’', '的情绪面标签为', sent, '概率为','%.2f%%' % round(prob * 100, 2))
if __name__ == '__main__':
# 测试
sentences = ['破烂平板', '手感不错,推荐购买', '刚开始吧还不错,但是后面越来越卡,差评',
'哈哈哈哈,我很喜欢', '今天很开心']
setiment(sentences)
from snownlp import SnowNLP # 调用情感分析函数
# 创建snownlp对象,设置要测试的语句
s1 = SnowNLP('这东西真的挺不错的')
s2 = SnowNLP('垃圾东西')
print('调用sentiments方法获取s1的积极情感概率为:',s1.sentiments)
print('调用sentiments方法获取s2的积极情感概率为:',s2.sentiments)

3、完成基于SnowNLP库的情感分析
from snownlp import SnowNLP # 调用情感分析函数
# 创建snownlp对象,设置要测试的语句
s1 = SnowNLP('这东西真的挺不错的')
s2 = SnowNLP('垃圾东西')
print('调用sentiments方法获取s1的积极情感概率为:',s1.sentiments)
print('调用sentiments方法获取s2的积极情感概率为:',s2.sentiments)

魔乐社区(Modelers.cn) 是一个中立、公益的人工智能社区,提供人工智能工具、模型、数据的托管、展示与应用协同服务,为人工智能开发及爱好者搭建开放的学习交流平台。社区通过理事会方式运作,由全产业链共同建设、共同运营、共同享有,推动国产AI生态繁荣发展。
更多推荐


所有评论(0)