-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathlearner.py
More file actions
110 lines (84 loc) · 3.68 KB
/
Copy pathlearner.py
File metadata and controls
110 lines (84 loc) · 3.68 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
# -*- coding: utf-8 -*-
import sys,json
from jubatus.classifier.client import Classifier
from jubatus.classifier.types import LabeledDatum
from jubatus.common import Datum
from tagger import CMUTweetTagger
from nltk.corpus import stopwords
from nltk.stem.porter import *
from neg_word.negation_cue import get_negation_cue
filter_tags = ['N', '^', 'V', 'A', 'R', '!', '#', 'E']
not_stem_tags = ['^', '!', '#', 'E']
negation_cues = get_negation_cue()
class sentimentLearner():
def __init__(self, model, happy_emoticons, sad_emoticons):
self.model = model
self.happy_emoticons = happy_emoticons
self.sad_emoticons = sad_emoticons
self.all_emoticons = map(lambda x:x.encode('utf-8') ,self.happy_emoticons + self.sad_emoticons)
self.stop = [word.encode('utf-8') for word in stopwords.words('english')]
self.stemmer = PorterStemmer()
#guess the label based on emoticon, train it if it can guess
#return boolean indicates whether it can be trained or not...
def trainTweet(self, tweet):
label = self.label(tweet)
if label:
tweet = self.processTweet(tweet)
print tweet
datum = Datum({"tweet": tweet})
labeldatum = LabeledDatum(label, datum)
self.model.train([labeldatum])
all_emoticons = self.happy_emoticons + self.sad_emoticons
for emoticon in all_emoticons:
tweet = tweet.replace(emoticon, "")
datum = Datum({"tweet": tweet})
labeldatum = LabeledDatum(label, datum)
self.model.train([labeldatum])
#save the model
self.model.save("sentimentModel")
return True
return False
#label the tweet based on emoticon
def label(self, tweet):
if any(emoticon in tweet for emoticon in self.happy_emoticons) and any(emoticon in tweet for emoticon in self.sad_emoticons):
return ""
elif any(emoticon in tweet for emoticon in self.happy_emoticons):
return "happy"
elif any(emoticon in tweet for emoticon in self.sad_emoticons):
return "sad"
return ""
def processTweet(self, tweet):
#tag the text
analyse = CMUTweetTagger.runtagger_parse([tweet])
analyse = analyse[0]
#only return these match the tag and not the stop words
filtered = filter(lambda x: (x[1] in filter_tags or x[0] in self.all_emoticons) and (x[0] not in self.stop), analyse)
#do not stem the emoticon
words = []
negation = False
# print filtered
for word_tag in filtered:
tag = word_tag[1]
word = word_tag[0]
if (tag in not_stem_tags or word in self.all_emoticons):
append_word = word
if (tag != 'E') and (word not in self.all_emoticons) and negation:
append_word = 'neg_' + append_word
words.append(word)
else:
#for all sorts of reason... might not able to stem the word...
try:
append_word = self.stemmer.stem(word.lower())
if negation:
append_word = 'neg_' + append_word
words.append( append_word.encode("utf-8") )
except:
# words.append(word[0])
continue
if (word.lower() in negation_cues) or word.lower().endswith("n't"):
negation = not negation
return " ".join(words).decode('utf-8')
def classify(self, tweet):
tweet = self.processTweet(tweet)
datum = Datum({"tweet": tweet})
print self.model.classify([datum])