-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathclassifyKaggle.py
More file actions
96 lines (75 loc) · 3.84 KB
/
Copy pathclassifyKaggle.py
File metadata and controls
96 lines (75 loc) · 3.84 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
'''
This program shell reads phrase data for the kaggle phrase sentiment classification problem.
The input to the program is the path to the kaggle directory "corpus" and a limit number.
The program reads all of the kaggle phrases, and then picks a random selection of the limit number.
It creates a "phrasedocs" variable with a list of phrases consisting of a pair
with the list of tokenized words from the phrase and the label number from 1 to 4
It prints a few example phrases.
In comments, it is shown how to get word lists from the two sentiment lexicons:
subjectivity and LIWC, if you want to use them in your features
Your task is to generate features sets and train and test a classifier.
Usage: python classifyKaggle.py <corpus directory path> <limit number>
'''
# open python and nltk packages needed for processing
import os
import sys
import random
import nltk
from nltk.corpus import stopwords
#import sentiment_read_subjectivity
# initialize the positive, neutral and negative word lists
#(positivelist, neutrallist, negativelist)= #sentiment_read_subjectivity.read_three_types('/Users/johnfields/Library/Mobile #Documents/com~apple~CloudDocs/Syracuse/IST664/Final Project/FinalProjectData/#kagglemoviereviews/SentimentLexicons/subjclueslen1-HLTEMNLP05.tff')
#SL= sentiment_read_subjectivity.readSubjectivity('/Users/johnfields/Library/Mobile #Documents/com~apple~CloudDocs/Syracuse/IST664/Final Project/FinalProjectData/#kagglemoviereviews/SentimentLexicons/subjclueslen1-HLTEMNLP05.tff')
import sentiment_read_LIWC_pos_neg_words
# initialize positve and negative word prefix lists from LIWC
# note there is another function isPresent to test if a word's prefix is in the list
(poslist, neglist) = sentiment_read_LIWC_pos_neg_words.read_words()
# define a feature definition function here
# use NLTK to compute evaluation measures from a reflist of gold labels
# and a testlist of predicted labels for all labels in a list
# returns lists of precision and recall for each label
# function to read kaggle training file, train and test a classifier
def processkaggle(dirPath,limitStr):
# convert the limit argument from a string to an int
limit = int(limitStr)
os.chdir(dirPath)
f = open('./train.tsv', 'r')
# loop over lines in the file and use the first limit of them
phrasedata = []
for line in f:
# ignore the first line starting with Phrase and read all lines
if (not line.startswith('Phrase')):
# remove final end of line character
line = line.strip()
# each line has 4 items separated by tabs
# ignore the phrase and sentence ids, and keep the phrase and sentiment
phrasedata.append(line.split('\t')[2:4])
# pick a random sample of length limit because of phrase overlapping sequences
random.shuffle(phrasedata)
phraselist = phrasedata[:limit]
print('Read', len(phrasedata), 'phrases, using', len(phraselist), 'random phrases')
for phrase in phraselist[:10]:
print (phrase)
# create list of phrase documents as (list of words, label)
phrasedocs = []
# add all the phrases
for phrase in phraselist:
tokens = nltk.word_tokenize(phrase[0])
phrasedocs.append((tokens, int(phrase[1])))
# print a few
for phrase in phrasedocs[:10]:
print (phrase)
# possibly filter tokens
# continue as usual to get all words and create word features
# feature sets from a feature definition function
# train classifier and show performance in cross-validation
"""
commandline interface takes a directory name with kaggle subdirectory for train.tsv
and a limit to the number of kaggle phrases to use
It then processes the files and trains a kaggle movie review sentiment classifier.
"""
if __name__ == '__main__':
if (len(sys.argv) != 3):
print ('usage: classifyKaggle.py <corpus-dir> <limit>')
sys.exit(0)
processkaggle(sys.argv[1], sys.argv[2])