forked from MiuLab/xSense
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathvisualize_online.py
More file actions
92 lines (83 loc) · 2.79 KB
/
Copy pathvisualize_online.py
File metadata and controls
92 lines (83 loc) · 2.79 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
#from multiprocessing import Queue
#from queue import PriorityQueue
from numpy import linalg as LA
from scipy import stats
import numpy as np
import argparse
import time
import sys
import os
from tqdm import tqdm
import pickle
top_k_words = []
zeros = 0.0
threshold = 0.001
h_dim = None
total = None
vectors = {}
num = 5
width = 10
def load_vectors(filename):
global vectors, dimensions, zeros, h_dim, total, top_k_words
vectors = {}
zeros = 0.0
f = open(filename, 'r')
lines = f.readlines()
f.close()
dimension = len(lines[0].split()) - 1
top_k_words = [ [] for i in range(dimension)]
c = 0
for line in tqdm(lines):
start = time.time()
words = line.strip().split()
vectors[words[0]] = [abs(float(i)) for i in words[1:]]
h_dim = len(words[1:])
c += 1
vector = vectors[words[0]]
for i, val in enumerate(vector):
temp = top_k_words[i]
if len(temp) < width:
temp.append((val,words[0]))
else:
check = temp[-1]
if check[0] < val:
temp[-1] = (val, words[0])
top_k_words[i] = sorted(temp, reverse=True)
zeros += sum([1 for i in vectors[words[0]] if i < threshold])
print ("Sparsity =", 100. * zeros/(len(lines)*dimension))
total = len(vectors)
print ('done loading vectors')
with open('vectors.pkl','wb') as f:
pickle.dump(vectors, f)
with open('top_k_words.pkl','wb') as f:
pickle.dump(top_k_words, f)
def load_top_dimensions(k):
global top_k_words
#return
dimensions = len(vectors[vectors.keys()[0]])
for i in range(dimensions):
temp = []
while top_k_words[i].qsize() > 0:
temp.append(top_k_words.get_nowait()[1])
top_k_words[i] = temp
print ('loaded top dimensions')
def find_top_participating_dimensions(word, k):
with open('vectors.pkl','rb') as f:
vectors = pickle.load(f)
with open('top_k_words.pkl','rb') as f:
top_k_words = pickle.load(f)
if word not in vectors:
print ('word not found')
return []
temp = [(j, i) for i, j in enumerate(vectors[word])]
answer = []
print (" -----------------------------------------------------")
print ("Word of interest = " , word)
for i, j in sorted(temp, reverse=True)[:k]:
print ("The contribution of the word '%s' in dimension %d = %f" %(word, j, i))
print ('Following are the top words in dimension', j, 'along with their contributions')
print (top_k_words[j])
#print('%s Top 10 closet word in dimension %d :'%(word, j))
#print([k[1] for k in top_k_words[j]])
answer.append([k[1] for k in top_k_words[j]])
return