Coverage for hopwise/evaluator/utils.py: 33%

60 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-09-30 13:25 +0000

1# @Time : 2020/08/04 

2# @Author : Kaiyuan Li 

3# @email : tsotfsk@outlook.com 

4 

5# UPDATE 

6# @Time : 2020/09/28, 2020/08/09 

7# @Author : Kaiyuan Li, Zhichao Feng 

8# @email : tsotfsk@outlook.com, fzcbupt@gmail.com 

9 

10"""hopwise.evaluator.utils 

11################################ 

12""" 

13 

14import itertools 

15from datetime import datetime 

16 

17import numpy as np 

18import pandas as pd 

19import torch 

20 

21 

22def pad_sequence(sequences, len_list, pad_to=None, padding_value=0): 

23 """Pad sequences to a matrix 

24 

25 Args: 

26 sequences (list): list of variable length sequences. 

27 len_list (list): the length of the tensors in the sequences 

28 pad_to (int, optional): if pad_to is not None, the sequences will pad to the length you set, 

29 else the sequence will pad to the max length of the sequences. 

30 padding_value (int, optional): value for padded elements. Default: 0. 

31 

32 Returns: 

33 torch.Tensor: [seq_num, max_len] or [seq_num, pad_to] 

34 

35 """ 

36 max_len = np.max(len_list) if pad_to is None else pad_to 

37 min_len = np.min(len_list) 

38 device = sequences[0].device 

39 if max_len == min_len: 

40 result = torch.cat(sequences, dim=0).view(-1, max_len) 

41 else: 

42 extra_len_list = np.subtract(max_len, len_list).tolist() 

43 padding_nums = max_len * len(len_list) - np.sum(len_list) 

44 padding_tensor = torch.tensor([-np.inf], device=device).repeat(padding_nums) 

45 padding_list = torch.split(padding_tensor, extra_len_list) 

46 result = list(itertools.chain.from_iterable(zip(sequences, padding_list))) 

47 result = torch.cat(result) 

48 

49 return result.view(-1, max_len) 

50 

51 

52def trunc(scores, method): 

53 """Round the scores by using the given method 

54 

55 Args: 

56 scores (numpy.ndarray): scores 

57 method (str): one of ['ceil', 'floor', 'around'] 

58 

59 Raises: 

60 NotImplementedError: method error 

61 

62 Returns: 

63 numpy.ndarray: processed scores 

64 """ 

65 try: 

66 cut_method = getattr(np, method) 

67 except NotImplementedError: 

68 raise NotImplementedError(f"module 'numpy' has no function named '{method}'") 

69 scores = cut_method(scores) 

70 return scores 

71 

72 

73def cutoff(scores, threshold): 

74 """Cut of the scores based on threshold 

75 

76 Args: 

77 scores (numpy.ndarray): scores 

78 threshold (float): between 0 and 1 

79 

80 Returns: 

81 numpy.ndarray: processed scores 

82 """ 

83 return np.where(scores > threshold, 1, 0) 

84 

85 

86def _binary_clf_curve(trues, preds): 

87 """Calculate true and false positives per binary classification threshold 

88 

89 Args: 

90 trues (numpy.ndarray): the true scores' list 

91 preds (numpy.ndarray): the predict scores' list 

92 

93 Returns: 

94 fps (numpy.ndarray): A count of false positives, at index i being the number of negative 

95 samples assigned a score >= thresholds[i] 

96 preds (numpy.ndarray): An increasing count of true positives, at index i being the number 

97 of positive samples assigned a score >= thresholds[i]. 

98 

99 Note: 

100 To improve efficiency, we referred to the source code(which is available at sklearn.metrics.roc_curve) 

101 in SkLearn and made some optimizations. 

102 

103 """ 

104 trues = trues == 1 

105 

106 desc_idxs = np.argsort(preds)[::-1] 

107 preds = preds[desc_idxs] 

108 trues = trues[desc_idxs] 

109 

110 unique_val_idxs = np.where(np.diff(preds))[0] 

111 threshold_idxs = np.r_[unique_val_idxs, trues.size - 1] 

112 

113 tps = np.cumsum(trues)[threshold_idxs] 

114 fps = 1 + threshold_idxs - tps 

115 return fps, tps 

116 

117 

118def plot_tsne_embeddings(model, **kwargs): 

119 import plotly.express as px 

120 

121 embeddings_list = list() 

122 identifiers_list = list() 

123 

124 for embeddings_name, embeddings in kwargs.items(): 

125 embeddings_list.append(embeddings) 

126 identifiers_list.extend([f"{embeddings_name} {id}" for id in range(embeddings.shape[0])]) 

127 

128 embeddings_list = np.concatenate(embeddings_list, axis=0) 

129 

130 combined_df = pd.DataFrame( 

131 { 

132 "x": embeddings_list[:, 0], 

133 "y": embeddings_list[:, 1], 

134 "type": [id.split(" ")[0] for id in identifiers_list], 

135 "identifier": identifiers_list, 

136 } 

137 ) 

138 

139 fig = px.scatter( 

140 combined_df, 

141 x="x", 

142 y="y", 

143 color="type", 

144 hover_data=["identifier"], 

145 labels={ 

146 "x": "Embedding Dimension 1", 

147 "y": "Embedding Dimension 2", 

148 }, 

149 title="Visualising Combined Embeddings", 

150 width=1024, 

151 height=1024, 

152 template="plotly_white", 

153 ) 

154 current_datetime = datetime.now() 

155 

156 fig.write_html(f"{model} tSNE {current_datetime.strftime('%b-%d-%Y_%H-%M-%S')}.html") 

157 

158 

159def train_tsne(model, config, load_best_model): 

160 from openTSNE import TSNE 

161 

162 tsne = TSNE( 

163 perplexity=config["perplexity"], 

164 n_jobs=config["n_jobs"], 

165 initialization=config["initialization"], 

166 metric=config["metric"], 

167 random_state=config["seed"], 

168 verbose=config["verbose"], 

169 ) 

170 

171 if ( 

172 (config["plot_on"] == "test" and load_best_model) 

173 or config["plot_on"] == "validation" 

174 or config["plot_on"] is not None 

175 ): 

176 try: 

177 tsne_user_embeddings = tsne.fit(model.user_embedding.weight.cpu().detach().numpy()) 

178 tsne_entity_embeddings = tsne.fit(model.entity_embedding.weight.cpu().detach().numpy()) 

179 tsne_relation_embeddings = tsne.fit(model.relation_embedding.weight.cpu().detach().numpy()) 

180 except AttributeError: 

181 print( 

182 "The model does not have the required embeddings for the t-SNE display, \ 

183 please check the name of the embeddings: user_embedding, entity_embedding, relation_embedding." 

184 ) 

185 

186 plot_tsne_embeddings( 

187 model=config["model"], 

188 user=tsne_user_embeddings, 

189 entity=tsne_entity_embeddings, 

190 relation=tsne_relation_embeddings, 

191 )