Coverage for hopwise/evaluator/utils.py: 33%
60 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 13:25 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 13:25 +0000
1# @Time : 2020/08/04
2# @Author : Kaiyuan Li
3# @email : tsotfsk@outlook.com
5# UPDATE
6# @Time : 2020/09/28, 2020/08/09
7# @Author : Kaiyuan Li, Zhichao Feng
8# @email : tsotfsk@outlook.com, fzcbupt@gmail.com
10"""hopwise.evaluator.utils
11################################
12"""
14import itertools
15from datetime import datetime
17import numpy as np
18import pandas as pd
19import torch
22def pad_sequence(sequences, len_list, pad_to=None, padding_value=0):
23 """Pad sequences to a matrix
25 Args:
26 sequences (list): list of variable length sequences.
27 len_list (list): the length of the tensors in the sequences
28 pad_to (int, optional): if pad_to is not None, the sequences will pad to the length you set,
29 else the sequence will pad to the max length of the sequences.
30 padding_value (int, optional): value for padded elements. Default: 0.
32 Returns:
33 torch.Tensor: [seq_num, max_len] or [seq_num, pad_to]
35 """
36 max_len = np.max(len_list) if pad_to is None else pad_to
37 min_len = np.min(len_list)
38 device = sequences[0].device
39 if max_len == min_len:
40 result = torch.cat(sequences, dim=0).view(-1, max_len)
41 else:
42 extra_len_list = np.subtract(max_len, len_list).tolist()
43 padding_nums = max_len * len(len_list) - np.sum(len_list)
44 padding_tensor = torch.tensor([-np.inf], device=device).repeat(padding_nums)
45 padding_list = torch.split(padding_tensor, extra_len_list)
46 result = list(itertools.chain.from_iterable(zip(sequences, padding_list)))
47 result = torch.cat(result)
49 return result.view(-1, max_len)
52def trunc(scores, method):
53 """Round the scores by using the given method
55 Args:
56 scores (numpy.ndarray): scores
57 method (str): one of ['ceil', 'floor', 'around']
59 Raises:
60 NotImplementedError: method error
62 Returns:
63 numpy.ndarray: processed scores
64 """
65 try:
66 cut_method = getattr(np, method)
67 except NotImplementedError:
68 raise NotImplementedError(f"module 'numpy' has no function named '{method}'")
69 scores = cut_method(scores)
70 return scores
73def cutoff(scores, threshold):
74 """Cut of the scores based on threshold
76 Args:
77 scores (numpy.ndarray): scores
78 threshold (float): between 0 and 1
80 Returns:
81 numpy.ndarray: processed scores
82 """
83 return np.where(scores > threshold, 1, 0)
86def _binary_clf_curve(trues, preds):
87 """Calculate true and false positives per binary classification threshold
89 Args:
90 trues (numpy.ndarray): the true scores' list
91 preds (numpy.ndarray): the predict scores' list
93 Returns:
94 fps (numpy.ndarray): A count of false positives, at index i being the number of negative
95 samples assigned a score >= thresholds[i]
96 preds (numpy.ndarray): An increasing count of true positives, at index i being the number
97 of positive samples assigned a score >= thresholds[i].
99 Note:
100 To improve efficiency, we referred to the source code(which is available at sklearn.metrics.roc_curve)
101 in SkLearn and made some optimizations.
103 """
104 trues = trues == 1
106 desc_idxs = np.argsort(preds)[::-1]
107 preds = preds[desc_idxs]
108 trues = trues[desc_idxs]
110 unique_val_idxs = np.where(np.diff(preds))[0]
111 threshold_idxs = np.r_[unique_val_idxs, trues.size - 1]
113 tps = np.cumsum(trues)[threshold_idxs]
114 fps = 1 + threshold_idxs - tps
115 return fps, tps
118def plot_tsne_embeddings(model, **kwargs):
119 import plotly.express as px
121 embeddings_list = list()
122 identifiers_list = list()
124 for embeddings_name, embeddings in kwargs.items():
125 embeddings_list.append(embeddings)
126 identifiers_list.extend([f"{embeddings_name} {id}" for id in range(embeddings.shape[0])])
128 embeddings_list = np.concatenate(embeddings_list, axis=0)
130 combined_df = pd.DataFrame(
131 {
132 "x": embeddings_list[:, 0],
133 "y": embeddings_list[:, 1],
134 "type": [id.split(" ")[0] for id in identifiers_list],
135 "identifier": identifiers_list,
136 }
137 )
139 fig = px.scatter(
140 combined_df,
141 x="x",
142 y="y",
143 color="type",
144 hover_data=["identifier"],
145 labels={
146 "x": "Embedding Dimension 1",
147 "y": "Embedding Dimension 2",
148 },
149 title="Visualising Combined Embeddings",
150 width=1024,
151 height=1024,
152 template="plotly_white",
153 )
154 current_datetime = datetime.now()
156 fig.write_html(f"{model} tSNE {current_datetime.strftime('%b-%d-%Y_%H-%M-%S')}.html")
159def train_tsne(model, config, load_best_model):
160 from openTSNE import TSNE
162 tsne = TSNE(
163 perplexity=config["perplexity"],
164 n_jobs=config["n_jobs"],
165 initialization=config["initialization"],
166 metric=config["metric"],
167 random_state=config["seed"],
168 verbose=config["verbose"],
169 )
171 if (
172 (config["plot_on"] == "test" and load_best_model)
173 or config["plot_on"] == "validation"
174 or config["plot_on"] is not None
175 ):
176 try:
177 tsne_user_embeddings = tsne.fit(model.user_embedding.weight.cpu().detach().numpy())
178 tsne_entity_embeddings = tsne.fit(model.entity_embedding.weight.cpu().detach().numpy())
179 tsne_relation_embeddings = tsne.fit(model.relation_embedding.weight.cpu().detach().numpy())
180 except AttributeError:
181 print(
182 "The model does not have the required embeddings for the t-SNE display, \
183 please check the name of the embeddings: user_embedding, entity_embedding, relation_embedding."
184 )
186 plot_tsne_embeddings(
187 model=config["model"],
188 user=tsne_user_embeddings,
189 entity=tsne_entity_embeddings,
190 relation=tsne_relation_embeddings,
191 )