-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathclassify.py
More file actions
146 lines (120 loc) · 3.57 KB
/
Copy pathclassify.py
File metadata and controls
146 lines (120 loc) · 3.57 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
#!/usr/bin/env python
import sys
import xml.dom.minidom
import re
import pickle
import nltk
import copy
import time
from nltk.corpus import stopwords
from nltk.stem.lancaster import LancasterStemmer
from nltk.stem.wordnet import WordNetLemmatizer
"""
This is to be run on the test data
"""
testdDoctlist = []
#fileOUT1='test_solution.out'
pattern=re.compile("[^\w']|_")
punctuation = re.compile(r'[-.?!,":;()|0-9]')
stopwords = stopwords.words('english')
K=20
lmtzr = WordNetLemmatizer()
def getText(nodelist):
rc = []
for node in nodelist:
if node.nodeType == node.TEXT_NODE:
rc.append(node.data)
return ''.join(rc)
def handleTok(tokenlist):
texts = ""
for token in tokenlist:
texts += " "+ getText(token.childNodes)
return texts
#get the matrix
matrixFile=open('matrix.txt','r')
matrix =pickle.load(matrixFile)
matrixFile.close()
def normalizeX(fdist):
sum1=0.0
for word in fdist:
sum1=sum1+(fdist.freq(word)**2)
sqr=(sum1**0.5)
return sqr
def calculateSim(doc,fdist1, xNorm):
sum=0.0
doc1 = copy.deepcopy(doc)
del doc1['observed category']
dNorm= doc1['normalized form']
del doc1['normalized form']
for keyWord in fdist1:
if keyWord in doc1:
sum=sum+(fdist1.freq(keyWord)*doc1[keyWord])
sim=0.0
if (xNorm* dNorm) != 0:
sim=(sum/(xNorm* dNorm))
else:
print "*****zero for doc %s"%doc1
return sim
if len(sys.argv) != 3:
print 'Usage: classify.py [path]input-filename output-filename'
sys.exit()
st = LancasterStemmer()
fileIN = sys.argv[1]
fileOUT1 =sys.argv[2]
f=open(fileIN,'r')
documentNumber=0
fout1=open(fileOUT1,'wb')
for rawDoc in f.readlines():
start = time.time()
fdist1 = nltk.FreqDist()
documentNumber+=1
rawDoc = '<root>' + rawDoc + '</root>'
dom = xml.dom.minidom.parseString(rawDoc)
#get the subject
subjectElement = dom.getElementsByTagName("subject")
subject = handleTok(subjectElement)
#get content
contentElement=dom.getElementsByTagName("content")
content = handleTok(contentElement)
allText=subject+ ' ' + content
allText=pattern.sub(' ', allText)
allText=' '.join(allText.split())
allText =punctuation.sub("", allText)
allText= allText.lower()
filtered_words = []
for w in str(allText).split():
if w not in stopwords:
w=st.stem(w)
w= lmtzr.lemmatize(w)
filtered_words.append(w)
fdist1.inc(w)
#doc is a row in the matrix
xNorm = normalizeX(fdist1)
similarityList=[]
matrow=0
ten3=0
for doc in matrix:
matrow+=1
docCategory=doc['observed category']
sim=calculateSim(doc, fdist1, xNorm)
similarityList.append({'similarity': sim, 'category': docCategory})
SortedsimilarityList=[]
SortedsimilarityList=sorted(similarityList, key=lambda k: k['similarity'],reverse=True)
categoryDist = nltk.FreqDist()
for i in xrange(K):
checkCategory=SortedsimilarityList[i]['category']
categoryDist.inc(checkCategory)
argmaxList = []
for category in categoryDist:
probOfCatDependDoc = (categoryDist[category]/K)
argmaxList.append({'probability': probOfCatDependDoc, 'category': category})
#theCategory=categoryDist.max()
theCategory=max(argmaxList, key=lambda k: k['probability'])['category'].strip()
fout1.write("%s\n" % theCategory)
fout1.flush()
#if documentNumber % 1000 == 0:
# ten3+=1
end = time.time()
docProcessingTime= end - start
print "processed %d docs. Processing time: %d (seconds)"%(documentNumber,docProcessingTime)
fout1.close()