Repository navigation
Expand file tree
/
Copy pathModifyTWords.py
More file actions
70 lines (47 loc) · 1.7 KB
/
Copy pathModifyTWords.py
File metadata and controls
70 lines (47 loc) · 1.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
__author__ = 'greglakomski'
import csv
# generates the more user readable words and codes file. Need to change filepaths for each experiment!
codehash = {}
filepath = "/Users/greglakomski/Desktop/GibbsLDA++-0.2/codesummary.csv"
with open(filepath) as f:
reader = csv.reader(f, delimiter=",")
for row in reader:
# create a hash table that just has the code and the short
# description to use is post processing results
codehash[row[0]]= row[1]
output = []
outputelement = []
filepath = "/Users/greglakomski/Desktop/GibbsLDA++-0.2/models/Medicare/FiveClustersStop/model-final.twords"
with open(filepath) as f:
reader = csv.reader(f, delimiter=" ")
for row in reader:
line = row[0].strip(' \t\n')
line2 = row[1].strip(' \t\n')
if line != 'Topic':
line = line.strip('[]')
line = line.strip("''")
line3 = row[3].strip(' \t\n')
#print line
#print(line +','+codehash[line])
outputelement.append(line)
if line in codehash:
#print 'ok'
outputelement.append(codehash[line])
outputelement.append(line3)
else:
#print 'not ok'
outputelement.append('CODE NOT FOUND')
output.append(outputelement)
outputelement= []
else:
#print('Topic'+ row[1])
outputelement.append('Topic '+ line2)
output.append(outputelement)
outputelement= []
# write the annotated word
filepath = "/Users/greglakomski/Desktop/GibbsLDA++-0.2/models/Medicare/FiveClustersStop/words_and_codes.csv"
with open(filepath,'w') as f:
writer = csv.writer(f, delimiter=',',quotechar=' ')
for x in output:
print(x)
writer.writerow(x)