-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathutilities.py
More file actions
198 lines (172 loc) · 8.67 KB
/
Copy pathutilities.py
File metadata and controls
198 lines (172 loc) · 8.67 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
import io
import numpy as np
import numpy.linalg as la
import pandas as pd
# Function from the FastText documentation
# https://fasttext.cc/docs/en/english-vectors.html
def load_vectors(fname):
"""
Load the pretrained fasttext vectors (.vec files).
:param fname: A string containing the file path to the model to load.
:return: a dictionary with each word as the key and its corresponding float vector as its value.
"""
fin = io.open(fname, 'r', encoding='utf-8', newline='\n', errors='ignore')
n, d = map(int, fin.readline().split())
data = {}
for line in fin:
tokens = line.rstrip().split(' ')
data[tokens[0]] = map(float, tokens[1:])
return data
def load_vectors_numpy(fname):
fin = io.open(fname, 'r', encoding='utf-8', newline='\n', errors='ignore')
n, d = map(int, fin.readline().split())
embeddings = np.zeros((n, d))
words = np.array([], dtype=object)
index = 0
for line in fin:
tokens = line.rstrip().split(' ')
words = np.append(words, tokens[0])
embeddings[index] = np.array(list(map(float, tokens[1:])))
index += 1
return words, embeddings
def get_array_word_embedding(words, embeddings, target):
index = np.where(words == target)
if len(index) > 0:
index = index[0]
return embeddings[index, :]
def get_unit_direction(p1, p2):
"""
Returns the normalized (length 1) unit vector of the direction between p1 and p2.
:param p1: the first word embedding represented as a vector in numpy
:param p2: the second word embedding represented as a vector in numpy;
Both vectors must have the same dimension.
:return: a unit vector in the direction of the difference between p1 and p2.
"""
assert(p1.shape == p2.shape)
return (p2 - p1) / la.norm(p2 - p1)
def get_projection(point, direction):
"""
Given a unit vector and a point, return the projection of that point onto the given direction.
:param point: a word embedding represented as a vector in numpy
:param direction: a unit directional vector of dimension d
:return: the projection of the word onto the direction, represented as a numpy vector
"""
return np.dot(point, direction) * direction
def get_point_distance(points, target):
"""
Get the distance between each point in an array and a target point.
:param points: a numpy array of length n or greater, containing points represented as vectors of dimension d
:param target: a point of dimension d to compare against other points, as a numpy array.
:return: a numpy array containing n scalars, representing the distance between each point and the target
"""
if len(points.shape) == 1:
return la.norm(points - target)
return la.norm(points - target, axis=1)
def get_midpoint(p1, p2):
"""
Given two points, finds the midpoint between those points in dimension d space.
:param p1: the first word embedding represented as a vector in numpy
:param p2: the second word embedding represented as a vector in numpy;
Both vectors must have the same dimension d.
:return: a point as a numpy array of dimension d.
"""
return p1 + ((p2 - p1) / 2)
def get_projection_matrix(points, direction):
"""
Get the projection of each word embedding onto the given direction.
:param points: a numpy array of length n or greater, containing points represented as vectors of dimension d
:param direction: a unit directional vector of dimension d
:return: a 2d numpy array of length n, containing projections for each word embedding.
"""
# A dot d gives a vector of scalars
# Dot product with the transpose of the direction to get the projections
# The direction should be a unit vector, so there is no need to normalize.
return np.outer(points.dot(direction), direction.T)
def get_orth_distance(point, direction):
"""
Get the orthogonal distance from a point to the subspace represented by the direction vector.
:param point: a word embedding represented as a vector of dimension d in numpy
:param direction: a unit directional vector of dimension d
:return: a scalar representing the distance from a word to the original direction vector
"""
return la.norm(point - get_projection(point, direction))
def get_orth_distance_matrix(points, direction):
"""
Get the orthogonal distance from collection of points to the subspace represented by the direction vector.
:param points: a numpy array of length n or greater, containing points represented as vectors of dimension d
:param direction: a unit directional vector of dimension d
:return: a numpy array of length n, containing the distance from each point to the original direction
"""
proj = (get_projection_matrix(points, direction))
return la.norm(points - proj, axis=1)
def search_by_distance(points, target, method="orth", filter="distance", threshold=10.0):
"""
A utility function for calculating various distances between a collection of words and some target vector.
The target vector may be a point or a unit direction vector.
:param points: a numpy array of length n or greater, containing points represented as vectors of dimension d
:param target: a vector of dimension d, represented as a numpy array
:param method: a string representing the kind of distance to calculate over the points.
If "orth", calculates the orthogonal distance between points and a direction vector.
If "point", calculates the Euclidean distance between points and a target point.
:param filter: a string representing how to filter the results once distances are calculated.
If None, returns results for every point.
If "distance", returns all results below a specified Euclidean distance threshold
If "quantile", returns all results below a specified distance quantile;
(in this case, the threshold should be between 0 and 1)
If "limit", returns the smallest k results
(in this case, the threshold should be a positive integer)
:param threshold: a value on which to filter the results of the distance calculation.
If the filter is None, this parameter is unused.
If "distance", this should be a positive continuous value representing Euclidean distance
If "quantile", this should be a value between 0 and 1, where 0.5 is the 50% quantile (the median)
If "limit", this should be a positive integer representing the smallest k values to return.
:return: distances: a numpy array containing the scalar distances from each point to the target
indices: returns the indices of the filtered words from the original array.
If filter is None, no indices are returned.
"""
if method is "orth":
distances = get_orth_distance_matrix(points, target)
elif method is "point":
distances = get_point_distance(points, target)
elif method is "proj":
distances = None
# TODO: want to find each point's projected distance from the midpoint
else:
print(f"Illegal distance type provided: {method}")
return None, None
if filter is None:
return distances, None
elif filter is "distance":
indices = distances <= threshold
return distances[indices], indices
elif filter is "quantile":
threshold = np.quantile(distances, threshold)
indices = distances <= threshold
return distances[indices], indices
elif filter is "limit":
quantile = threshold / len(points)
threshold = np.quantile(distances, quantile)
indices = distances <= threshold
return distances[indices], indices
else:
print(f"Illegal search method provided: {filter}")
def flatten_ft_predictions(results):
labels, probabilities = results
flat_labels = [x[0] for x in labels]
flat_probs = np.array(probabilities).flatten()
return flat_labels, flat_probs
def find_prediction_disagreement(inputs, predictions, compare_baseline=False):
is_diff = predictions[0] == predictions[0]
if compare_baseline:
for i in range(1, len(predictions)):
is_diff = ((predictions[0] == predictions[i]) & is_diff)
is_diff = ~is_diff
else:
for i in range(len(predictions) - 1):
for j in range(i+1, len(predictions)):
is_diff = ((~(predictions[i] == predictions[j])) & is_diff)
if type(inputs) is list:
disagreements = pd.Series(inputs)[is_diff]
else:
disagreements = inputs[is_diff]
return disagreements, is_diff