Coverage for hopwise/evaluator/metrics.py: 89%
689 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 13:25 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-09-30 13:25 +0000
1# @Time : 2020/08/04
2# @Author : Kaiyuan Li
3# @email : tsotfsk@outlook.com
5# UPDATE
6# @Time : 2020/08/12, 2021/8/29, 2020/9/16, 2021/7/2
7# @Author : Kaiyuan Li, Zhichao Feng, Xingyu Pan, Zihan Lin
8# @email : tsotfsk@outlook.com, fzcbupt@gmail.com, panxy@ruc.edu.cn, zhlin@ruc.edu.cn
10# UPDATE
11# @Time : 2025
12# @Author : Giacomo Medda, Alessandro Soccol
13# @Email : giacomo.medda@unica.it, alessandro.soccol@unica.it
15r"""hopwise.evaluator.metrics
16############################
18Suppose there is a set of :math:`n` items to be ranked. Given a user :math:`u` in the user set :math:`U`,
19we use :math:`\hat R(u)` to represent a ranked list of items that a model produces, and :math:`R(u)` to
20represent a ground-truth set of items that user :math:`u` has interacted with. For top-k recommendation, only
21top-ranked items are important to consider. Therefore, in top-k evaluation scenarios, we truncate the
22recommendation list with a length :math:`K`. Besides, in loss-based metrics, :math:`S` represents the
23set of user(u)-item(i) pairs, :math:`\hat r_{u i}` represents the score predicted by the model,
24:math:`{r}_{u i}` represents the ground-truth labels.
26"""
28import inspect
29import sys
30from collections import Counter
31from logging import getLogger
33import numpy as np
34from sklearn.metrics import auc as sk_auc
35from sklearn.metrics import mean_absolute_error, mean_squared_error
37from hopwise.evaluator.base_metric import AbstractMetric, ConsumerTopKMetric, LossMetric, PathQualityMetric, TopkMetric
38from hopwise.evaluator.utils import _binary_clf_curve
39from hopwise.utils import EvaluatorType
41# TopK Metrics
44class Hit(TopkMetric):
45 r"""HR_ (also known as truncated Hit-Ratio) is a way of calculating how many 'hits'
46 you have in an n-sized list of ranked items. If there is at least one item that falls in the ground-truth set,
47 we call it a hit.
49 .. _HR: https://medium.com/@rishabhbhatia315/recommendation-system-evaluation-metrics-3f6739288870
51 .. math::
52 \mathrm {HR@K} = \frac{1}{|U|}\sum_{u \in U} \delta(\hat{R}(u) \cap R(u) \neq \emptyset),
54 :math:`\delta(·)` is an indicator function. :math:`\delta(b)` = 1 if :math:`b` is true and 0 otherwise.
55 :math:`\emptyset` denotes the empty set.
56 """
58 def __init__(self, config):
59 super().__init__(config)
61 def calculate_metric(self, dataobject):
62 pos_index, _ = self.used_info(dataobject)
63 result = self.metric_info(pos_index)
64 metric_dict = self.topk_result("hit", result)
65 return metric_dict
67 def metric_info(self, pos_index):
68 result = np.cumsum(pos_index, axis=1)
69 return (result > 0).astype(int)
72class MRR(TopkMetric):
73 r"""The MRR_ (also known as Mean Reciprocal Rank) computes the reciprocal rank
74 of the first relevant item found by an algorithm.
76 .. _MRR: https://en.wikipedia.org/wiki/Mean_reciprocal_rank
78 .. math::
79 \mathrm {MRR@K} = \frac{1}{|U|}\sum_{u \in U} \frac{1}{\operatorname{rank}_{u}^{*}}
81 :math:`{rank}_{u}^{*}` is the rank position of the first relevant item found by an algorithm for a user :math:`u`.
82 """
84 def __init__(self, config):
85 super().__init__(config)
87 def calculate_metric(self, dataobject):
88 pos_index, _ = self.used_info(dataobject)
89 result = self.metric_info(pos_index)
90 metric_dict = self.topk_result("mrr", result)
91 return metric_dict
93 def metric_info(self, pos_index):
94 idxs = pos_index.argmax(axis=1)
95 result = np.zeros_like(pos_index, dtype=np.float64)
96 for row, idx in enumerate(idxs):
97 if pos_index[row, idx] > 0:
98 result[row, idx:] = 1 / (idx + 1)
99 else:
100 result[row, idx:] = 0
101 return result
104class MAP(TopkMetric):
105 r"""MAP_ (also known as Mean Average Precision) is meant to calculate
106 average precision for the relevant items.
108 Note:
109 In this case the normalization factor used is :math:`\frac{1}{min(|\hat R(u)|, K)}`, which prevents your
110 AP score from being unfairly suppressed when your number of recommendations couldn't possibly capture
111 all the correct ones.
113 .. _MAP: http://sdsawtelle.github.io/blog/output/mean-average-precision-MAP-for-recommender-systems.html#MAP-for-Recommender-Algorithms
115 .. math::
116 \mathrm{MAP@K} = \frac{1}{|U|}\sum_{u \in U} (\frac{1}{min(|\hat R(u)|, K)} \sum_{j=1}^{|\hat{R}(u)|} I\left(\hat{R}_{j}(u) \in R(u)\right) \cdot Precision@j)
118 :math:`\hat{R}_{j}(u)` is the j-th item in the recommendation list of \hat R (u)).
119 """ # noqa: E501
121 def __init__(self, config):
122 super().__init__(config)
123 self.config = config
125 def calculate_metric(self, dataobject):
126 pos_index, pos_len = self.used_info(dataobject)
127 result = self.metric_info(pos_index, pos_len)
128 metric_dict = self.topk_result("map", result)
129 return metric_dict
131 def metric_info(self, pos_index, pos_len):
132 pre = pos_index.cumsum(axis=1) / np.arange(1, pos_index.shape[1] + 1)
133 sum_pre = np.cumsum(pre * pos_index.astype(np.float64), axis=1)
134 len_rank = np.full_like(pos_len, pos_index.shape[1])
135 actual_len = np.where(pos_len > len_rank, len_rank, pos_len)
136 result = np.zeros_like(pos_index, dtype=np.float64)
137 for row, lens in enumerate(actual_len):
138 ranges = np.arange(1, pos_index.shape[1] + 1)
139 ranges[lens:] = ranges[lens - 1]
140 result[row] = sum_pre[row] / ranges
141 return result
144class Recall(TopkMetric):
145 r"""Recall_ is a measure for computing the fraction of relevant items out of all relevant items.
147 .. _recall: https://en.wikipedia.org/wiki/Precision_and_recall#Recall
149 .. math::
150 \mathrm {Recall@K} = \frac{1}{|U|}\sum_{u \in U} \frac{|\hat{R}(u) \cap R(u)|}{|R(u)|}
152 :math:`|R(u)|` represents the item count of :math:`R(u)`.
153 """
155 def __init__(self, config):
156 super().__init__(config)
158 def calculate_metric(self, dataobject):
159 pos_index, pos_len = self.used_info(dataobject)
160 result = self.metric_info(pos_index, pos_len)
161 metric_dict = self.topk_result("recall", result)
162 return metric_dict
164 def metric_info(self, pos_index, pos_len):
165 return np.cumsum(pos_index, axis=1) / pos_len.reshape(-1, 1)
168class NDCG(TopkMetric):
169 r"""NDCG_ (also known as normalized discounted cumulative gain) is a measure of ranking quality,
170 where positions are discounted logarithmically. It accounts for the position of the hit by assigning
171 higher scores to hits at top ranks.
173 .. _NDCG: https://en.wikipedia.org/wiki/Discounted_cumulative_gain#Normalized_DCG
175 .. math::
176 \mathrm {NDCG@K} = \frac{1}{|U|}\sum_{u \in U} (\frac{1}{\sum_{i=1}^{\min (|R(u)|, K)}
177 \frac{1}{\log _{2}(i+1)}} \sum_{i=1}^{K} \delta(i \in R(u)) \frac{1}{\log _{2}(i+1)})
179 :math:`\delta(·)` is an indicator function.
180 """
182 def __init__(self, config):
183 super().__init__(config)
185 def calculate_metric(self, dataobject):
186 pos_index, pos_len = self.used_info(dataobject)
187 result = self.metric_info(pos_index, pos_len)
188 metric_dict = self.topk_result("ndcg", result)
189 return metric_dict
191 def metric_info(self, pos_index, pos_len):
192 len_rank = np.full_like(pos_len, pos_index.shape[1])
193 idcg_len = np.where(pos_len > len_rank, len_rank, pos_len)
195 iranks = np.zeros_like(pos_index, dtype=np.float64)
196 iranks[:, :] = np.arange(1, pos_index.shape[1] + 1)
197 idcg = np.cumsum(1.0 / np.log2(iranks + 1), axis=1)
198 for row, idx in enumerate(idcg_len):
199 idcg[row, idx:] = idcg[row, idx - 1]
201 ranks = np.zeros_like(pos_index, dtype=np.float64)
202 ranks[:, :] = np.arange(1, pos_index.shape[1] + 1)
203 dcg = 1.0 / np.log2(ranks + 1)
204 dcg = np.cumsum(np.where(pos_index, dcg, 0), axis=1)
206 result = dcg / idcg
207 return result
210class Precision(TopkMetric):
211 r"""Precision_ (also called positive predictive value) is a measure for computing the fraction of relevant items
212 out of all the recommended items. We average the metric for each user :math:`u` get the final result.
214 .. _precision: https://en.wikipedia.org/wiki/Precision_and_recall#Precision
216 .. math::
217 \mathrm {Precision@K} = \frac{1}{|U|}\sum_{u \in U} \frac{|\hat{R}(u) \cap R(u)|}{|\hat {R}(u)|}
219 :math:`|\hat R(u)|` represents the item count of :math:`\hat R(u)`.
220 """
222 def __init__(self, config):
223 super().__init__(config)
225 def calculate_metric(self, dataobject):
226 pos_index, _ = self.used_info(dataobject)
227 result = self.metric_info(pos_index)
228 metric_dict = self.topk_result("precision", result)
229 return metric_dict
231 def metric_info(self, pos_index):
232 return pos_index.cumsum(axis=1) / np.arange(1, pos_index.shape[1] + 1)
235# CTR Metrics
238class GAUC(AbstractMetric):
239 r"""GAUC (also known as Grouped Area Under Curve) is used to evaluate the two-class model, referring to
240 the area under the ROC curve grouped by user. We weighted the index of each user :math:`u` by the number of positive
241 samples of users to get the final result.
243 For further details, please refer to the `paper <https://dl.acm.org/doi/10.1145/3219819.3219823>`__
245 Note:
246 It calculates the AUC score of each user, and finally obtains GAUC by weighting the user AUC.
247 It is also not limited to k. Due to our padding for `scores_tensor` with `-np.inf`, the padding
248 value will influence the ranks of origin items. Therefore, we use descending sort here and make
249 an identity transformation to the formula of `AUC`, which is shown in `auc_` function.
250 For readability, we didn't do simplification in the code.
252 .. math::
253 \begin{align*}
254 \mathrm {AUC(u)} &= \frac {{{|R(u)|} \times {(n+1)} - \frac{|R(u)| \times (|R(u)|+1)}{2}} -
255 \sum\limits_{i=1}^{|R(u)|} rank_{i}} {{|R(u)|} \times {(n - |R(u)|)}} \\
256 \mathrm{GAUC} &= \frac{1}{\sum_{u \in U} |R(u)|}\sum_{u \in U} |R(u)| \cdot(\mathrm {AUC(u)})
257 \end{align*}
259 :math:`rank_i` is the descending rank of the i-th items in :math:`R(u)`.
260 """ # noqa: E501
262 metric_type = EvaluatorType.RANKING
263 metric_need = ["rec.meanrank"]
265 def __init__(self, config):
266 super().__init__(config)
268 def calculate_metric(self, dataobject):
269 mean_rank = dataobject.get("rec.meanrank").numpy()
270 pos_rank_sum, user_len_list, pos_len_list = np.split(mean_rank, 3, axis=1)
271 user_len_list, pos_len_list = (
272 user_len_list.squeeze(-1),
273 pos_len_list.squeeze(-1),
274 )
275 result = self.metric_info(pos_rank_sum, user_len_list, pos_len_list)
276 return {"gauc": round(result, self.decimal_place)}
278 def metric_info(self, pos_rank_sum, user_len_list, pos_len_list):
279 """Get the value of GAUC metric.
281 Args:
282 pos_rank_sum (numpy.ndarray): sum of descending rankings for positive items of each users.
283 user_len_list (numpy.ndarray): the number of predicted items for users.
284 pos_len_list (numpy.ndarray): the number of positive items for users.
286 Returns:
287 float: The value of the GAUC.
288 """
289 neg_len_list = user_len_list - pos_len_list
290 # check positive and negative samples
291 any_without_pos = np.any(pos_len_list == 0)
292 any_without_neg = np.any(neg_len_list == 0)
293 non_zero_idx = np.full(len(user_len_list), True, dtype=np.bool)
294 if any_without_pos:
295 logger = getLogger()
296 logger.warning(
297 "No positive samples in some users, "
298 "true positive value should be meaningless, "
299 "these users have been removed from GAUC calculation"
300 )
301 non_zero_idx *= pos_len_list != 0
302 if any_without_neg:
303 logger = getLogger()
304 logger.warning(
305 "No negative samples in some users, "
306 "false positive value should be meaningless, "
307 "these users have been removed from GAUC calculation"
308 )
309 non_zero_idx *= neg_len_list != 0
310 if any_without_pos or any_without_neg:
311 item_list = user_len_list, neg_len_list, pos_len_list, pos_rank_sum
312 user_len_list, neg_len_list, pos_len_list, pos_rank_sum = map(lambda x: x[non_zero_idx], item_list)
314 pair_num = (
315 (user_len_list + 1) * pos_len_list - pos_len_list * (pos_len_list + 1) / 2 - np.squeeze(pos_rank_sum)
316 )
317 user_auc = pair_num / (neg_len_list * pos_len_list)
318 result = (user_auc * pos_len_list).sum() / pos_len_list.sum()
319 return result
322class AUC(LossMetric):
323 r"""AUC_ (also known as Area Under Curve) is used to evaluate the two-class model, referring to
324 the area under the ROC curve.
326 .. _AUC: https://en.wikipedia.org/wiki/Receiver_operating_characteristic#Area_under_the_curve
328 Note:
329 This metric does not calculate group-based AUC which considers the AUC scores
330 averaged across users. It is also not limited to k. Instead, it calculates the
331 scores on the entire prediction results regardless the users. We call the interface
332 in `scikit-learn`, and code calculates the metric using the variation of following formula.
334 .. math::
335 \mathrm {AUC} = \frac {{{M} \times {(N+1)} - \frac{M \times (M+1)}{2}} -
336 \sum\limits_{i=1}^{M} rank_{i}} {{M} \times {(N - M)}}
338 :math:`M` denotes the number of positive items.
339 :math:`N` denotes the total number of user-item interactions.
340 :math:`rank_i` denotes the descending rank of the i-th positive item.
341 """
343 def __init__(self, config):
344 super().__init__(config)
346 def calculate_metric(self, dataobject):
347 return self.output_metric("auc", dataobject)
349 def metric_info(self, preds, trues):
350 fps, tps = _binary_clf_curve(trues, preds)
351 if len(fps) > 2: # noqa: PLR2004
352 optimal_idxs = np.where(np.r_[True, np.logical_or(np.diff(fps, 2), np.diff(tps, 2)), True])[0]
353 fps = fps[optimal_idxs]
354 tps = tps[optimal_idxs]
356 tps = np.r_[0, tps]
357 fps = np.r_[0, fps]
359 if fps[-1] <= 0:
360 logger = getLogger()
361 logger.warning("No negative samples in y_true, false positive value should be meaningless")
362 fpr = np.repeat(np.nan, fps.shape)
363 else:
364 fpr = fps / fps[-1]
366 if tps[-1] <= 0:
367 logger = getLogger()
368 logger.warning("No positive samples in y_true, true positive value should be meaningless")
369 tpr = np.repeat(np.nan, tps.shape)
370 else:
371 tpr = tps / tps[-1]
373 result = sk_auc(fpr, tpr)
374 return result
377# Loss-based Metrics
380class MAE(LossMetric):
381 r"""MAE_ (also known as Mean Absolute Error regression loss) is used to evaluate the difference between
382 the score predicted by the model and the actual behavior of the user.
384 .. _MAE: https://en.wikipedia.org/wiki/Mean_absolute_error
386 .. math::
387 \mathrm{MAE}=\frac{1}{|{S}|} \sum_{(u, i) \in {S}}\left|\hat{r}_{u i}-r_{u i}\right|
389 :math:`|S|` represents the number of pairs in :math:`S`.
390 """
392 smaller = True
394 def __init__(self, config):
395 super().__init__(config)
397 def calculate_metric(self, dataobject):
398 return self.output_metric("mae", dataobject)
400 def metric_info(self, preds, trues):
401 return mean_absolute_error(trues, preds)
404class RMSE(LossMetric):
405 r"""RMSE_ (also known as Root Mean Squared Error) is another error metric like `MAE`.
407 .. _RMSE: https://en.wikipedia.org/wiki/Root-mean-square_deviation
409 .. math::
410 \mathrm{RMSE} = \sqrt{\frac{1}{|{S}|} \sum_{(u, i) \in {S}}(\hat{r}_{u i}-r_{u i})^{2}}
411 """
413 smaller = True
415 def __init__(self, config):
416 super().__init__(config)
418 def calculate_metric(self, dataobject):
419 return self.output_metric("rmse", dataobject)
421 def metric_info(self, preds, trues):
422 return np.sqrt(mean_squared_error(trues, preds))
425class LogLoss(LossMetric):
426 r"""Logloss_ (also known as logistic loss or cross-entropy loss) is used to evaluate the probabilistic
427 output of the two-class classifier.
429 .. _Logloss: http://wiki.fast.ai/index.php/Log_Loss
431 .. math::
432 LogLoss = \frac{1}{|S|} \sum_{(u,i) \in S}(-((r_{u i} \ \log{\hat{r}_{u i}}) + {(1 - r_{u i})}\ \log{(1 - \hat{r}_{u i})}))
433 """ # noqa: E501
435 smaller = True
437 def __init__(self, config):
438 super().__init__(config)
440 def calculate_metric(self, dataobject):
441 return self.output_metric("logloss", dataobject)
443 def metric_info(self, preds, trues):
444 eps = 1e-15
445 preds = np.float64(preds)
446 preds = np.clip(preds, eps, 1 - eps)
447 loss = np.sum(-trues * np.log(preds) - (1 - trues) * np.log(1 - preds))
448 return loss / len(preds)
451class ItemCoverage(AbstractMetric):
452 r"""ItemCoverage_ computes the coverage of recommended items over all items.
454 .. _ItemCoverage: https://en.wikipedia.org/wiki/Coverage_(information_systems)
456 For further details, please refer to the `paper <https://dl.acm.org/doi/10.1145/1864708.1864761>`__
457 and `paper <https://link.springer.com/article/10.1007/s13042-017-0762-9>`__.
459 .. math::
460 \mathrm{Coverage@K}=\frac{\left| \bigcup_{u \in U} \hat{R}(u) \right|}{|I|}
461 """
463 metric_type = EvaluatorType.RANKING
464 metric_need = ["rec.items", "data.num_items"]
466 def __init__(self, config):
467 super().__init__(config)
468 self.topk = config["topk"]
470 def used_info(self, dataobject):
471 """Get the matrix of recommendation items and number of items in total item set"""
472 item_matrix = dataobject.get("rec.items")
473 num_items = dataobject.get("data.num_items")
474 return item_matrix.numpy(), num_items
476 def calculate_metric(self, dataobject):
477 item_matrix, num_items = self.used_info(dataobject)
478 metric_dict = {}
479 for k in self.topk:
480 key = "{}@{}".format("itemcoverage", k)
481 metric_dict[key] = round(self.get_coverage(item_matrix[:, :k], num_items), self.decimal_place)
482 return metric_dict
484 def get_coverage(self, item_matrix, num_items):
485 """Get the coverage of recommended items over all items
487 Args:
488 item_matrix(numpy.ndarray): matrix of items recommended to users.
489 num_items(int): the total number of items.
491 Returns:
492 float: the `coverage` metric.
493 """
494 unique_count = np.unique(item_matrix).shape[0]
495 return unique_count / num_items
498class AveragePopularity(AbstractMetric):
499 r"""AveragePopularity computes the average popularity of recommended items.
501 For further details, please refer to the `paper <https://arxiv.org/abs/1205.6700>`__
502 and `paper <https://link.springer.com/article/10.1007/s13042-017-0762-9>`__.
504 .. math::
505 \mathrm{AveragePopularity@K}=\frac{1}{|U|} \sum_{u \in U } \frac{\sum_{i \in R_{u}} \phi(i)}{|R_{u}|}
507 :math:`\phi(i)` is the number of interaction of item i in training data.
508 """
510 metric_type = EvaluatorType.RANKING
511 smaller = True
512 metric_need = ["rec.items", "data.count_items"]
514 def __init__(self, config):
515 super().__init__(config)
516 self.topk = config["topk"]
518 def used_info(self, dataobject):
519 """Get the matrix of recommendation items and the popularity of items in training data"""
520 item_counter = dataobject.get("data.count_items")
521 item_matrix = dataobject.get("rec.items")
522 return item_matrix.numpy(), dict(item_counter)
524 def calculate_metric(self, dataobject):
525 item_matrix, item_count = self.used_info(dataobject)
526 result = self.metric_info(self.get_pop(item_matrix, item_count))
527 metric_dict = self.topk_result("averagepopularity", result)
528 return metric_dict
530 def get_pop(self, item_matrix, item_count):
531 """Convert the matrix of item id to the matrix of item popularity using a dict:{id,count}.
533 Args:
534 item_matrix(numpy.ndarray): matrix of items recommended to users.
535 item_count(dict): the number of interaction of items in training data.
537 Returns:
538 numpy.ndarray: the popularity of items in the recommended list.
539 """
540 value = np.zeros_like(item_matrix)
541 for i in range(item_matrix.shape[0]):
542 row = item_matrix[i, :]
543 for j in range(row.shape[0]):
544 value[i][j] = item_count.get(row[j], 0)
545 return value
547 def metric_info(self, values):
548 return values.cumsum(axis=1) / np.arange(1, values.shape[1] + 1)
550 def topk_result(self, metric, value):
551 """Match the metric value to the `k` and put them in `dictionary` form
553 Args:
554 metric(str): the name of calculated metric.
555 value(numpy.ndarray): metrics for each user, including values from `metric@1` to `metric@max(self.topk)`.
557 Returns:
558 dict: metric values required in the configuration.
559 """
560 metric_dict = {}
561 avg_result = value.mean(axis=0)
562 for k in self.topk:
563 key = f"{metric}@{k}"
564 metric_dict[key] = round(avg_result[k - 1], self.decimal_place)
565 return metric_dict
568class ShannonEntropy(AbstractMetric):
569 r"""ShannonEntropy_ presents the diversity of the recommendation items.
570 It is the entropy over items' distribution.
572 .. _ShannonEntropy: https://en.wikipedia.org/wiki/Entropy_(information_theory)
574 For further details, please refer to the `paper <https://arxiv.org/abs/1205.6700>`__
575 and `paper <https://link.springer.com/article/10.1007/s13042-017-0762-9>`__
577 .. math::
578 \mathrm {ShannonEntropy@K}=-\sum_{i=1}^{|I|} p(i) \log p(i)
580 :math:`p(i)` is the probability of recommending item i
581 which is the number of item i in recommended list over all items.
582 """
584 metric_type = EvaluatorType.RANKING
585 metric_need = ["rec.items"]
587 def __init__(self, config):
588 super().__init__(config)
589 self.topk = config["topk"]
591 def used_info(self, dataobject):
592 """Get the matrix of recommendation items."""
593 item_matrix = dataobject.get("rec.items")
594 return item_matrix.numpy()
596 def calculate_metric(self, dataobject):
597 item_matrix = self.used_info(dataobject)
598 metric_dict = {}
599 for k in self.topk:
600 key = "{}@{}".format("shannonentropy", k)
601 metric_dict[key] = round(self.get_entropy(item_matrix[:, :k]), self.decimal_place)
602 return metric_dict
604 def get_entropy(self, item_matrix):
605 """Get shannon entropy through the top-k recommendation list.
607 Args:
608 item_matrix(numpy.ndarray): matrix of items recommended to users.
610 Returns:
611 float: the shannon entropy.
612 """
613 item_count = dict(Counter(item_matrix.flatten()))
614 total_num = item_matrix.shape[0] * item_matrix.shape[1]
615 result = 0.0
616 for cnt in item_count.values():
617 p = cnt / total_num
618 result += -p * np.log(p)
619 return result / len(item_count)
622class GiniIndex(AbstractMetric):
623 r"""GiniIndex presents the diversity of the recommendation items.
624 It is used to measure the inequality of a distribution.
626 .. _GiniIndex: https://en.wikipedia.org/wiki/Gini_coefficient
628 For further details, please refer to the `paper <https://dl.acm.org/doi/10.1145/3308560.3317303>`__.
630 .. math::
631 \mathrm {GiniIndex@K}=\left(\frac{\sum_{i=1}^{|I|}(2 i-|I|-1) P{(i)}}{|I| \sum_{i=1}^{|I|} P{(i)}}\right)
633 :math:`P{(i)}` represents the number of times all items appearing in the recommended list,
634 which is indexed in non-decreasing order (P_{(i)} \leq P_{(i+1)}).
635 """
637 metric_type = EvaluatorType.RANKING
638 smaller = True
639 metric_need = ["rec.items", "data.num_items"]
641 def __init__(self, config):
642 super().__init__(config)
643 self.topk = config["topk"]
645 def used_info(self, dataobject):
646 """Get the matrix of recommendation items and number of items in total item set"""
647 item_matrix = dataobject.get("rec.items")
648 num_items = dataobject.get("data.num_items")
649 return item_matrix.numpy(), num_items
651 def calculate_metric(self, dataobject):
652 item_matrix, num_items = self.used_info(dataobject)
653 metric_dict = {}
654 for k in self.topk:
655 key = "{}@{}".format("giniindex", k)
656 metric_dict[key] = round(self.get_gini(item_matrix[:, :k], num_items), self.decimal_place)
657 return metric_dict
659 def get_gini(self, item_matrix, num_items):
660 """Get gini index through the top-k recommendation list.
662 Args:
663 item_matrix(numpy.ndarray): matrix of items recommended to users.
664 num_items(int): the total number of items.
666 Returns:
667 float: the gini index.
668 """
669 item_count = dict(Counter(item_matrix.flatten()))
670 sorted_count = np.array(sorted(item_count.values()))
671 num_recommended_items = sorted_count.shape[0]
672 total_num = item_matrix.shape[0] * item_matrix.shape[1]
673 idx = np.arange(num_items - num_recommended_items + 1, num_items + 1)
674 gini_index = np.sum((2 * idx - num_items - 1) * sorted_count) / total_num
675 gini_index /= num_items
676 return gini_index
679class TailPercentage(AbstractMetric):
680 r"""TailPercentage_ computes the percentage of long-tail items in recommendation items.
682 .. _TailPercentage: https://en.wikipedia.org/wiki/Long_tail#Criticisms
684 For further details, please refer to the `paper <https://arxiv.org/pdf/2007.12329.pdf>`__.
686 .. math::
687 \mathrm {TailPercentage@K}=\frac{1}{|U|} \sum_{u \in U} \frac{\sum_{i \in R_{u}} {\delta(i \in T)}}{|R_{u}|}
689 :math:`\delta(·)` is an indicator function.
690 :math:`T` is the set of long-tail items,
691 which is a portion of items that appear in training data seldomly.
693 Note:
694 If you want to use this metric, please set the parameter 'tail_ratio' in the config
695 which can be an integer or a float in (0,1]. Otherwise it will default to 0.1.
696 """
698 metric_type = EvaluatorType.RANKING
699 metric_need = ["rec.items", "data.count_items"]
701 def __init__(self, config):
702 super().__init__(config)
703 self.topk = config["topk"]
704 self.tail = config["tail_ratio"]
705 if self.tail is None or self.tail <= 0:
706 self.tail = 0.1
708 def used_info(self, dataobject):
709 """Get the matrix of recommendation items and number of items in total item set."""
710 item_matrix = dataobject.get("rec.items")
711 count_items = dataobject.get("data.count_items")
712 return item_matrix.numpy(), dict(count_items)
714 def get_tail(self, item_matrix, count_items):
715 """Get long-tail percentage through the top-k recommendation list.
717 Args:
718 item_matrix(numpy.ndarray): matrix of items recommended to users.
719 count_items(dict): the number of interaction of items in training data.
721 Returns:
722 float: long-tail percentage.
723 """
724 if self.tail > 1:
725 tail_items = [item for item, cnt in count_items.items() if cnt <= self.tail]
726 else:
727 count_items = sorted(count_items.items(), key=lambda kv: (kv[1], kv[0]))
728 cut = max(int(len(count_items) * self.tail), 1)
729 count_items = count_items[:cut]
730 tail_items = [item for item, cnt in count_items]
731 value = np.zeros_like(item_matrix)
732 for i in range(item_matrix.shape[0]):
733 row = item_matrix[i, :]
734 for j in range(row.shape[0]):
735 value[i][j] = 1 if row[j] in tail_items else 0
736 return value
738 def calculate_metric(self, dataobject):
739 item_matrix, count_items = self.used_info(dataobject)
740 result = self.metric_info(self.get_tail(item_matrix, count_items))
741 metric_dict = self.topk_result("tailpercentage", result)
742 return metric_dict
744 def metric_info(self, values):
745 return values.cumsum(axis=1) / np.arange(1, values.shape[1] + 1)
747 def topk_result(self, metric, value):
748 """Match the metric value to the `k` and put them in `dictionary` form.
750 Args:
751 metric(str): the name of calculated metric.
752 value(numpy.ndarray): metrics for each user, including values from `metric@1` to `metric@max(self.topk)`.
754 Returns:
755 dict: metric values required in the configuration.
756 """
757 metric_dict = {}
758 avg_result = value.mean(axis=0)
759 for k in self.topk:
760 key = f"{metric}@{k}"
761 metric_dict[key] = round(avg_result[k - 1], self.decimal_place)
762 return metric_dict
765# Consumer Metrics dynamic creation
768def create_consumer_metric_class(topk_metric):
769 """Dynamically creates and returns a new consumer class with the given name, base classes, and attributes."""
770 topk_metric_class = getattr(sys.modules[__name__], topk_metric)
771 signature = inspect.signature(topk_metric_class.metric_info)
772 parameters = signature.parameters
774 if "pos_len" in parameters:
776 def ranking_metric_info(self, pos_index, pos_len):
777 return self.ranking_metric.metric_info(pos_index, pos_len)
778 else:
780 def ranking_metric_info(self, pos_index, pos_len):
781 return self.ranking_metric.metric_info(pos_index)
783 consumer_metric_class_name = f"Delta{topk_metric}"
784 consumer_metric_class = type(
785 consumer_metric_class_name, (ConsumerTopKMetric,), {"ranking_metric_info": ranking_metric_info}
786 )
787 consumer_metric_class.metric_need.extend(
788 [need for need in topk_metric_class.metric_need if need not in consumer_metric_class.metric_need]
789 )
791 def factory_init(self, config):
792 super(consumer_metric_class, self).__init__(config)
793 self.ranking_metric = topk_metric_class(config)
795 consumer_metric_class.__init__ = factory_init
797 return consumer_metric_class
800DeltaHit = create_consumer_metric_class("Hit")
801DeltaMRR = create_consumer_metric_class("MRR")
802DeltaMAP = create_consumer_metric_class("MAP")
803DeltaNDCG = create_consumer_metric_class("NDCG")
804DeltaPrecision = create_consumer_metric_class("Precision")
805DeltaRecall = create_consumer_metric_class("Recall")
807# Beyond Accuracy Metrics
810class Serendipity(AbstractMetric):
811 r"""Serendipity is a ranking-based metric that measures how *unexpected* the recommended
812 items are to a user with respect to item popularity in the training data.
814 The intuition behind serendipity is that recommendations are more serendipitous
815 when they avoid globally popular items and instead promote items that the user
816 is unlikely to encounter by chance.
818 Note:
819 This implementation defines serendipity as a *popularity-based unexpectedness*
820 measure. For each user, it compares the recommended top-k items with the most
821 popular items in the training data (excluding items already interacted with
822 by the user).
824 The metric is computed per user and per rank position, then averaged across
825 users. It is explicitly dependent on `k`, and different values of `k` may
826 produce different serendipity scores.
828 In particular:
829 - Item popularity is derived from the training interaction counts.
830 - Items appearing in the user history are excluded from the popularity ranking.
831 - A recommendation is considered *non-serendipitous* if it belongs to the
832 top-k most popular items.
834 Formally, for a user :math:`u`, let:
835 - :math:`R_u = (r_{u,1}, \dots, r_{u,k})` be the ranked list of recommended items,
836 - :math:`P_u^k` be the set of the top-k most popular items after removing items
837 already interacted with by user :math:`u`.
839 The serendipity at cutoff :math:`k` is defined as:
841 .. math::
842 \mathrm{Serendipity@k}(u) =
843 1 - \frac{1}{k} \sum_{i=1}^{k} \mathbb{I}\left[r_{u,i} \in P_u^k\right]
845 where :math:`\mathbb{I}[\cdot]` is the indicator function.
847 The final metric value is obtained by averaging over all users:
849 .. math::
850 \mathrm{Serendipity@k} = \frac{1}{|U|} \sum_{u \in U} \mathrm{Serendipity@k}(u)
852 Higher values indicate more serendipitous recommendations, while lower values
853 indicate recommendations dominated by popular items.
854 """
856 metric_type = EvaluatorType.RANKING
857 metric_need = ["rec.items", "data.num_items", "data.num_users", "data.count_items", "data.history_index"]
859 def __init__(self, config):
860 super().__init__(config)
861 self.topk = config["topk"]
863 def used_info(self, dataobject):
864 """Get the matrix of recommendation items and the popularity of items in training data"""
865 num_users = dataobject.get("data.num_users")
866 num_items = dataobject.get("data.num_items")
867 item_counter = dataobject.get("data.count_items")
868 item_matrix = dataobject.get("rec.items")
869 history_matrix = dataobject.get("data.history_index")
871 items, count = zip(*item_counter.items())
872 item_counter = np.zeros(num_items, dtype=int)
873 item_counter[list(items)] = count
875 return item_matrix.numpy(), item_counter, history_matrix, num_users
877 def get_popularity_rank(self, item_count, history_matrix, num_users):
878 """Rank every item by decreasing popularity, separately for each user.
880 Items a user already interacted with and the padding item are pushed below every item with a valid count,
881 items without training interactions included, so they never displace a candidate from the popular set.
883 Args:
884 item_count(numpy.ndarray): number of interactions of each item in training data.
885 history_matrix(numpy.ndarray): user and item indices of the training interactions.
886 num_users(int): number of users, padding included.
888 Returns:
889 numpy.ndarray: popularity rank of every item, shape of ``(n_users, n_items)``, 0 being the most popular.
890 """
891 pop_recs = np.tile(item_count, (num_users, 1))
892 pop_recs[tuple(history_matrix)] = -1
893 pop_recs[:, 0] = -1 # padding item
894 pop_recs = pop_recs[1:] # remove the padding user
896 pop_order = np.argsort(pop_recs, axis=-1)[:, ::-1]
897 pop_rank = np.empty_like(pop_order)
898 np.put_along_axis(pop_rank, pop_order, np.arange(pop_recs.shape[1]), axis=1)
899 return pop_rank
901 def metric_info(self, item_matrix, pop_rank):
902 """Look up the popularity rank of each recommended item.
904 A recommended item is among the `k` most popular ones iff its rank is lower than `k`,
905 so this single matrix serves every cutoff and no set intersection is needed.
907 Returns:
908 numpy.ndarray: popularity rank of the recommended items, shape of ``(n_users, max(topk))``.
909 """
910 return np.take_along_axis(pop_rank, item_matrix, axis=1)
912 def calculate_metric(self, dataobject):
913 item_matrix, item_count, history_matrix, num_users = self.used_info(dataobject)
915 pop_rank = self.get_popularity_rank(item_count, history_matrix, num_users)
916 result = self.metric_info(item_matrix, pop_rank)
917 metric_dict = self.topk_result("serendipity", result)
918 return metric_dict
920 def topk_result(self, metric, value):
921 """Match the metric value to the `k` and put them in `dictionary` form.
923 Args:
924 metric(str): the name of calculated metric.
925 value(numpy.ndarray): popularity rank of the recommended items, shape of ``(n_users, max(topk))``.
927 Returns:
928 dict: metric values required in the configuration.
929 """
930 metric_dict = {}
931 for k in self.topk:
932 key = f"{metric}@{k}"
933 popular_topk = (value[:, :k] < k).sum(axis=1)
934 metric_dict[key] = round((1 - popular_topk / k).mean(), self.decimal_place)
935 return metric_dict
938class Novelty(AbstractMetric):
939 r"""Novelty is a ranking-based metric that measures how *unpopular* the recommended
940 items are with respect to their popularity in the training data.
942 The intuition behind novelty is that recommendations are more novel when they
943 promote items that are rarely interacted with, rather than frequently consumed
944 popular items.
946 Note:
947 This implementation defines novelty as the inverse of normalized item popularity.
948 Item popularity is computed from the interaction counts observed in the training
949 data and then normalized using min-max normalization.
951 The metric is computed at different cutoffs `k`, and the final value for each
952 `k` is obtained by averaging the novelty scores across users.
954 In particular:
955 - Item popularity is derived from training interaction frequencies.
956 - Popularity values are normalized globally using min-max normalization.
957 - Novelty is computed as the inverse of normalized popularity.
958 - The metric is explicitly dependent on `k`.
960 Formally, let:
961 - :math:`R_u^k = (r_{u,1}, \dots, r_{u,k})` be the top-k items recommended to user :math:`u`,
962 - :math:`c(i)` be the interaction count of item :math:`i` in the training data,
963 - :math:`c_{\min}` and :math:`c_{\max}` be the minimum and maximum item popularity,
964 respectively.
966 The normalized popularity of an item :math:`i` is defined as:
968 .. math::
969 \hat{c}(i) = \frac{c(i) - c_{\min}}{c_{\max} - c_{\min}}
971 The novelty of an item :math:`i` is then:
973 .. math::
974 \mathrm{novelty}(i) = 1 - \hat{c}(i)
976 The novelty score for a user at cutoff :math:`k` is computed as:
978 .. math::
979 \mathrm{Novelty@k}(u) = \frac{1}{k} \sum_{i \in R_u^k} \left(1 - \hat{c}(i)\right)
981 Finally, the overall novelty is obtained by averaging across all users:
983 .. math::
984 \mathrm{Novelty@k} = \frac{1}{|U|} \sum_{u \in U} \mathrm{Novelty@k}(u)
986 Higher values indicate more novel recommendations, favoring items with lower
987 training popularity.
988 """
990 metric_type = EvaluatorType.RANKING
991 metric_need = ["rec.items", "data.count_items", "data.num_items"]
993 def __init__(self, config):
994 super().__init__(config)
995 self.topk = config["topk"]
997 def used_info(self, dataobject):
998 """Get the matrix of recommendation items and number of items in total item set"""
999 item_matrix = dataobject.get("rec.items")
1000 count_items = dataobject.get("data.count_items")
1001 num_items = dataobject.get("data.num_items")
1002 return item_matrix.numpy(), dict(count_items), int(num_items)
1004 def metric_info(self, item_count, num_items):
1005 counts = np.zeros(num_items)
1006 counts[list(item_count.keys())] = list(item_count.values())
1008 min_pop = min(item_count.values()) if len(item_count.values()) == num_items else 0
1009 max_pop = max(item_count.values())
1011 item_novelty = 1 - (counts - min_pop) / (max_pop - min_pop)
1013 return item_novelty
1015 def calculate_metric(self, dataobject):
1016 item_matrix, item_count, num_items = self.used_info(dataobject)
1017 item_novelty = self.metric_info(item_count, num_items)
1018 per_rank = item_novelty[item_matrix].cumsum(axis=1) / np.arange(1, item_matrix.shape[1] + 1)
1019 avg_result = per_rank.mean(axis=0)
1021 metric_dict = {}
1022 for k in self.topk:
1023 metric_dict[f"novelty@{k}"] = round(avg_result[k - 1], self.decimal_place)
1024 return metric_dict
1027# Perceived Path Explanation Quality
1030class Fidelity(PathQualityMetric):
1031 r"""Fidelity (FID) is an explanation quality metric that measures the proportion of
1032 recommended items that can be explained by at least one explanation path.
1034 Note:
1035 In this implementation, an item is considered *explainable* for a user if there exists
1036 at least one explanation path connecting the user to that item.
1037 Fidelity is computed at cutoff :math:`k` and averaged across users.
1039 This definition follows the hopwise paper, where fidelity is described as the
1040 *percentage of recommended items that are explainable*.
1042 Formally, let:
1043 - :math:`U` be the set of users,
1044 - :math:`E_u^k` be the set of top-k recommended items for user :math:`u` that admit
1045 at least one explanation path.
1047 The fidelity at cutoff :math:`k` is defined as:
1049 .. math::
1050 \mathrm{FID@k}
1051 =
1052 \frac{1}{|U|}
1053 \sum_{u \in U}
1054 \min\!\left(
1055 \frac{|E_u^k|}{k},
1056 1
1057 \right)
1059 Higher values indicate that a larger fraction of the recommended items is covered
1060 by explanations.
1061 """
1063 def __init__(self, config):
1064 super().__init__(config)
1065 self.topk = config["topk"]
1067 def calculate_metric(self, dataobject):
1068 paths = self.used_info(dataobject)
1069 result = self.metric_info(paths)
1070 metric_dict = self.topk_result("Fidelity", result)
1071 return metric_dict
1073 def metric_info(self, paths):
1074 user_paths = dict()
1075 for user, item, _, _ in paths:
1076 if user not in user_paths:
1077 user_paths[user] = set()
1078 user_paths[user].add(item)
1080 path_topk_len = []
1081 for items in user_paths.values():
1082 path_topk_len.append(len(items))
1084 return np.array(path_topk_len)
1086 def topk_result(self, metric, value):
1087 """Match the metric value to the `k` and put them in `dictionary` form.
1089 Args:
1090 metric(str): the name of calculated metric.
1091 value(numpy.ndarray): metrics for each user, including values from `metric@1` to
1092 `metric@max(self.topk)`.
1094 Returns:
1095 dict: metric values required in the configuration.
1096 """
1097 metric_dict = {}
1098 for k in self.topk:
1099 key = f"{metric}@{k}"
1100 avg_result = (value / k).clip(min=0.0, max=1.0).mean(axis=0)
1101 metric_dict[key] = round(avg_result, self.decimal_place)
1102 return metric_dict
1105class LIR(PathQualityMetric):
1106 r"""Linking Interaction Recency (LIR)
1108 This property serves to quantify the time since the linking interaction in the explanation path occurred.
1109 Given a user :math:`u \in U` and the set :math:`P_u` of items this user interacted with,
1110 we denote the list of their interactions, sorted chronologically, by :math:`T_u = [(p^i, t^i)]`,
1111 where :math:`p^i \in P_u` is a item experienced by the user, :math:`t^i \in \mathbb N`
1112 is the timestamp that interaction occurred, and :math:`t^i \leq t^{i+1}` :math:`\forall i = 1, \dots, |P_u|`.
1114 They applied an exponentially weighed moving average to the timestamps included in :math:`T_u`,
1115 to obtain the LIR of each interaction performed by the user :math:`u`.
1116 Specifically, given an interaction :math:`(p^i, t^i) \in T_u`, the LIR for that interaction
1117 was computed as follows:
1119 .. math::
1120 \mathrm{LIR(p^i, t^i)} = ( 1 - \beta_{LIR} ) \cdot LIR(p^{i-1}, t^{i-1}) + \beta_{LIR} \cdot t^i
1123 where :math:`\beta_{LIR} \in [0, 1]` is a decay associated to the interaction time,
1124 and :math:`LIR(p^1, t^1) = t^1`.
1125 The LIR values were min-max normalized for each user to lay in the range :math:`[0, 1]`,
1126 with values close to 0 (1) meaning that the linking interaction is far away (recent) in time.
1127 In the case of a recommended item not being present in the user's interaction history,
1128 the LIR of the linking interaction was set to 0, meaning that the linking interaction
1129 is very recent (i.e., the user has just interacted with the item).
1130 The overall LIR for explanations in a recommended list was obtained
1131 by averaging the LIR of the linking interactions for the selected explanation path of each recommended item.
1133 For further details, please refer to the `paper <https://dl.acm.org/doi/pdf/10.1145/3477495.3532041>`.
1134 """
1136 metric_need = ["data.timestamp", "data.num_items"]
1138 def __init__(self, config):
1139 super().__init__(config)
1140 self.li_idx = 1 # The linking interaction is the second element in the path tuple
1142 def calculate_metric(self, dataobject):
1143 paths = self.used_info(dataobject)
1144 timestamp_matrix = dataobject.get("data.timestamp")
1145 num_items = dataobject.get("data.num_items")
1147 lir_matrix = self.get_lir_matrix(timestamp_matrix)
1148 result = self.metric_info(lir_matrix, paths, num_items)
1149 metric_dict = self.topk_result("lir", result)
1150 return metric_dict
1152 def get_lir_matrix(self, timestamp_matrix):
1153 lir_matrix = np.zeros_like(timestamp_matrix, dtype=np.float32)
1154 for uid, user_inter_timestamp in enumerate(timestamp_matrix):
1155 inter_mask = user_inter_timestamp > 0
1156 if not inter_mask.any():
1157 continue
1159 sort_idxs = np.argsort(user_inter_timestamp[inter_mask])
1160 sorted_timestamps = user_inter_timestamp[inter_mask][sort_idxs]
1161 ema_timestamps = self.normalized_ema(sorted_timestamps)
1162 if not np.isnan(ema_timestamps).any():
1163 inter_indices = np.where(inter_mask)[0][sort_idxs]
1164 lir_matrix[uid, inter_indices] = ema_timestamps
1166 return lir_matrix
1168 def metric_info(self, lir_matrix, paths, num_items):
1169 users, li_items = [], []
1170 for user, _, _, path in paths:
1171 is_item = path[self.li_idx][1] == "item" or (
1172 path[self.li_idx][1] == "entity" and path[self.li_idx][-1] < num_items
1173 )
1174 if is_item:
1175 users.append(user)
1176 li_items.append(int(path[self.li_idx][-1]))
1178 return lir_matrix[users, li_items]
1181class SEP(PathQualityMetric):
1182 r"""Popularity of Shared Entity (SEP)
1184 This property serves to quantify the extent to which the shared entity included in an explanation-path is popular.
1185 They assume that the number of relationships a shared entity is involved in the KG is a proxy of its popularity.
1186 For instance, the popularity of an actor is computed by counting how many movies that actor starred in.
1187 They denote the list of entities of a given type $\lambda$ in the KG, sorted based on their popularity,
1188 by :math:`_\lambda = [(e^i, v^i)]`, where :math:`e^i \in E_\lambda` is an entity
1189 of type :math:`\lambda`, :math:`v^i \in \mathbb N` is the number of relationships a shared entity is involved in
1190 (in-degree),
1191 and :math:`v^i \leq v^{i+1}` :math:`\forall i = 1, \dots, |E_\lambda|`.
1192 We applied an exponential decay to the popularity scores in :math:`S_\lambda$, to get the SEP of an entity
1193 of type :math:`\lambda`, as:
1195 .. math::
1196 \mathrm{SEP(e^i, v^i)} = ( 1 - \beta_{SEP} ) \cdot SEP(e^{i-1}, v^{i-1}) + \beta_{SEP} \cdot v^i
1198 where :math:`\beta_{SEP}` is a decay related to the popularity, and :math:`SEP(e^1, v^1) = v^1`
1199 The SEP values w ere min-max normalized for each entity type to lay in the range [0, 1],
1200 with values close to 0 (1) when the entity has a low (high) popularity.
1201 The shared entity novelty might be obtained as :math:`SEN(e^i, v^i) = 1-SEP(e^i, v^i).
1203 The overall SEP for explanations in a recommended list was obtained by averaging the SEP
1204 values of the shared entity # noqa: E501
1205 in the selected path for each recommended item.
1207 For further details, please refer to the `paper <https://dl.acm.org/doi/pdf/10.1145/3477495.3532041>`.
1208 """
1210 metric_need = ["data.node_degree"]
1212 def __init__(self, config):
1213 super().__init__(config)
1215 def calculate_metric(self, dataobject):
1216 paths = self.used_info(dataobject)
1217 node_degree = dataobject.get("data.node_degree")
1219 sep_matrix = self.get_sep_matrix(node_degree)
1220 result = self.metric_info(paths, sep_matrix)
1221 metric_dict = self.topk_result("SEP", result)
1222 return metric_dict
1224 def get_sep_matrix(self, node_degree):
1225 # Precompute entity distribution
1226 sep_matrix = {}
1228 for node_type, eid_degree in node_degree.items():
1229 eid_degree_tuples = list(zip(eid_degree.keys(), eid_degree.values()))
1230 eid_degree_tuples.sort(key=lambda x: x[1])
1231 ema_es = self.normalized_ema([x[1] for x in eid_degree_tuples])
1232 iid_weight = {}
1233 for idx in range(len(ema_es)):
1234 iid = eid_degree_tuples[idx][0]
1235 iid_weight[iid] = ema_es[idx]
1237 sep_matrix[node_type] = iid_weight
1238 return sep_matrix
1240 def metric_info(self, paths, sep_matrix):
1241 seps_topk = []
1242 for _, _, _, path in paths:
1243 shared_entity_id, shared_entity_type = path[-2][-1], path[-2][1]
1244 if shared_entity_type == "item":
1245 shared_entity_type = "entity"
1246 seps_topk.append(sep_matrix[shared_entity_type][shared_entity_id])
1247 if not seps_topk:
1248 return np.array([])
1249 return np.array(seps_topk)
1252class LID(PathQualityMetric):
1253 r"""LID (Linked Interaction Diversity) is a path quality metric that measures the
1254 diversity of linking interactions used to generate explanations for recommendations.
1256 Note:
1257 In this implementation, LID evaluates how many *distinct linking interactions*
1258 (i.e., intermediate items or entities) are involved in a user's explanation paths.
1259 The metric is computed per user as the ratio between the number of distinct
1260 linking interaction identifiers and the total number of explanation paths,
1261 and then averaged across users.
1263 LID is independent of the cutoff :math:`k`. The same value is reported for all
1264 configured values of `k`, consistently with the current hopwise implementation.
1266 Formally, let:
1267 - :math:`U` be the set of users,
1268 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1269 - :math:`LI_u` be the set of distinct linking interaction identifiers appearing
1270 as the second node of the paths in :math:`P_u`.
1272 The LID metric is defined as:
1274 .. math::
1275 \mathrm{LID}
1276 =
1277 \frac{1}{|U|}
1278 \sum_{u \in U}
1279 \frac{|LI_u|}{|P_u|}
1281 Higher values indicate that a larger variety of linking interactions is used to
1282 explain the recommendations, reflecting more diverse explanatory structures.
1283 """
1285 def __init__(self, config):
1286 super().__init__(config)
1288 def calculate_metric(self, dataobject):
1289 paths = self.used_info(dataobject)
1290 result = self.metric_info(paths)
1291 metric_dict = self.topk_result("LID", result)
1292 return metric_dict
1294 def metric_info(self, paths):
1295 unique_linking_interaction = dict()
1296 for user, _, _, path in paths:
1297 linked_interaction_id = path[1][-1]
1298 if user not in unique_linking_interaction:
1299 unique_linking_interaction[user] = [0, set()]
1300 unique_linking_interaction[user][0] += 1 # number of path for that specific user
1301 unique_linking_interaction[user][1].add(linked_interaction_id)
1303 lid = []
1304 for user in unique_linking_interaction:
1305 n_user_paths = unique_linking_interaction[user][0]
1306 li = unique_linking_interaction[user][1]
1307 if not n_user_paths:
1308 lid.append(0.0)
1309 continue
1310 lid.append(len(li) / n_user_paths)
1312 return np.array(lid)
1315class SED(PathQualityMetric):
1316 r"""SED (Shared Entity Diversity) is a path quality metric that measures the
1317 diversity of shared entities involved in explanation paths.
1319 Note:
1320 In this implementation, SED evaluates how many *distinct shared entities*
1321 are used to explain recommendations for each user. The shared entity is
1322 defined as the penultimate node of an explanation path.
1323 The metric is computed per user as the ratio between the number of distinct
1324 shared entity identifiers and the total number of explanation paths, and then
1325 averaged across users.
1327 SED is independent of the cutoff :math:`k`. The same value is reported for all
1328 configured values of `k`, consistently with the current hopwise implementation.
1330 Formally, let:
1331 - :math:`U` be the set of users,
1332 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1333 - :math:`SE_u` be the set of distinct shared entity identifiers appearing
1334 as the penultimate node of the paths in :math:`P_u`.
1336 The SED metric is defined as:
1338 .. math::
1339 \mathrm{SED}
1340 =
1341 \frac{1}{|U|}
1342 \sum_{u \in U}
1343 \frac{|SE_u|}{|P_u|}
1345 Higher values indicate that explanations rely on a more diverse set of shared
1346 entities, reflecting richer and less repetitive explanatory structures.
1347 """
1349 def __init__(self, config):
1350 super().__init__(config)
1352 def calculate_metric(self, dataobject):
1353 paths = self.used_info(dataobject)
1355 result = self.metric_info(paths)
1356 metric_dict = self.topk_result("SED", result)
1357 return metric_dict
1359 def metric_info(self, paths):
1360 unique_shared_entity = dict()
1361 for user, _, _, path in paths:
1362 shared_entity_id = path[-2][-1]
1363 if user not in unique_shared_entity:
1364 unique_shared_entity[user] = [0, set()]
1365 # number of path for that specific user
1366 unique_shared_entity[user][0] += 1
1367 unique_shared_entity[user][1].add(shared_entity_id)
1369 sed = []
1370 for user in unique_shared_entity:
1371 n_user_paths = unique_shared_entity[user][0]
1372 se = unique_shared_entity[user][1]
1373 if not n_user_paths:
1374 sed.append(0.0)
1375 continue
1376 sed.append(len(se) / n_user_paths)
1378 return np.array(list(sed))
1381class PTD(PathQualityMetric):
1382 r"""PTD (Path Type Diversity) is a path quality metric that measures the
1383 diversity of path types used in explanation paths.
1385 Note:
1386 In this implementation, PTD evaluates how many *distinct path types* are
1387 involved in a user's explanation paths. The path type is identified by the
1388 first element of the last node in the path; if the last node corresponds to
1389 a self-loop, the path type is taken from the penultimate node.
1391 The metric is computed per user as the ratio between the number of distinct
1392 path types and the minimum between the number of explanation paths and the
1393 total number of possible path types. The final score is obtained by averaging
1394 across users.
1396 PTD is independent of the cutoff :math:`k`. The same value is reported for all
1397 configured values of `k`, consistently with the current hopwise implementation.
1399 Formally, let:
1400 - :math:`U` be the set of users,
1401 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1402 - :math:`T_u` be the set of distinct path types appearing in :math:`P_u`,
1403 - :math:`T` be the set of all possible path types.
1405 The PTD metric is defined as:
1407 .. math::
1408 \mathrm{PTD}
1409 =
1410 \frac{1}{|U|}
1411 \sum_{u \in U}
1412 \frac{|T_u|}{\min\left(|P_u|,\;|T|\right)}
1414 Higher values indicate that explanations exploit a wider variety of path
1415 types, reflecting more structurally diverse explanation patterns."""
1417 metric_need = ["data.max_path_type"]
1419 def __init__(self, config):
1420 super().__init__(config)
1422 def calculate_metric(self, dataobject):
1423 paths = self.used_info(dataobject)
1424 max_path_type = dataobject.get("data.max_path_type")
1426 result = self.metric_info(paths, max_path_type)
1427 metric_dict = self.topk_result("PTD", result)
1428 return metric_dict
1430 def metric_info(self, paths, max_path_type):
1431 unique_path_type = dict()
1432 for user, _, _, path in paths:
1433 # Get path type
1434 path_type = path[-1][0]
1435 if path_type == "self_loop": # Handle size 3
1436 path_type = path[-2][0]
1437 # Track number of paths for each user and seen path types
1438 if user not in unique_path_type:
1439 unique_path_type[user] = [0, set()]
1440 # number of path for that specific user
1441 unique_path_type[user][0] += 1
1442 unique_path_type[user][1].add(path_type)
1444 ptd = []
1445 for user in unique_path_type:
1446 n_user_paths = unique_path_type[user][0]
1447 pt = unique_path_type[user][1]
1448 if not n_user_paths:
1449 ptd.append(0.0)
1450 continue
1451 ptd.append(len(pt) / min(n_user_paths, len(max_path_type)))
1453 return np.array(ptd)
1456class PTC(PathQualityMetric):
1457 r"""PTC (Path Type Concentration) is a path quality metric that measures how evenly
1458 explanation paths are spread across path types.
1460 Note:
1461 Despite its name, PTC is reported as the complement of a Simpson concentration
1462 index (i.e., the Gini-Simpson diversity index): for each user, it is one minus the
1463 probability that two explanation paths drawn without replacement share the same
1464 path type. Higher values therefore mean lower concentration.
1465 The path type is identified by the first element of the last node in the path;
1466 if the last node corresponds to a self-loop, the path type is taken from the
1467 penultimate node. Users with a single explanation path get a PTC of 0.
1469 The metric is computed per user and then averaged across users.
1470 PTC is independent of the cutoff :math:`k`; the same value is reported for all
1471 configured values of `k`, consistently with the current hopwise implementation.
1473 Formally, let:
1474 - :math:`U` be the set of users,
1475 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1476 - :math:`N_u = |P_u|` be the number of paths for user :math:`u`,
1477 - :math:`n_{u,t}` be the number of paths of type :math:`t` for user :math:`u`.
1479 The PTC metric is defined as:
1481 .. math::
1482 \mathrm{PTC}
1483 =
1484 \frac{1}{|U|}
1485 \sum_{u \in U}
1486 \left(
1487 1 - \frac{\sum_{t} n_{u,t}(n_{u,t}-1)}{N_u(N_u-1)}
1488 \right)
1490 Higher values indicate that explanation paths are distributed across multiple
1491 path types (lower concentration), while lower values indicate that explanations
1492 are dominated by a small number of path types.
1493 """
1495 metric_need = ["data.max_path_type"]
1497 def __init__(self, config):
1498 super().__init__(config)
1500 def calculate_metric(self, dataobject):
1501 paths = self.used_info(dataobject)
1502 max_path_type = dataobject.get("data.max_path_type")
1504 result = self.metric_info(paths, max_path_type)
1505 metric_dict = self.topk_result("PTC", result)
1506 return metric_dict
1508 def metric_info(self, paths, max_path_type):
1509 user_simpson_index = dict()
1510 for user, _, _, path in paths:
1511 if user not in user_simpson_index:
1512 # 0 is N, the second is n_path_for_patterns
1513 user_simpson_index[user] = [0, {k: 0 for k in set(max_path_type)}]
1514 # Get path type
1515 path_type = path[-1][0]
1517 if path_type == "self_loop": # Handle size 3
1518 path_type = path[-2][0]
1520 if path_type not in user_simpson_index[user][1]:
1521 user_simpson_index[user][1][path_type] = 0
1523 user_simpson_index[user][1][path_type] += 1
1524 user_simpson_index[user][0] += 1
1526 ptc = []
1527 for user in user_simpson_index:
1528 numerator = 0
1529 for path_type, n_path_type_ith in user_simpson_index[user][1].items():
1530 numerator += n_path_type_ith * (n_path_type_ith - 1)
1532 if user_simpson_index[user][0] * (user_simpson_index[user][0] - 1) == 0:
1533 ptc.append(0)
1534 continue
1536 ptc.append(1 - (numerator / (user_simpson_index[user][0] * (user_simpson_index[user][0] - 1))))
1538 return np.array(ptc)
1541class PPT(PathQualityMetric):
1542 r"""PPT (Path Pattern Type) is a path quality metric that measures the diversity
1543 of relational path patterns used in explanation paths.
1545 Note:
1546 In this implementation, a path pattern is defined as the ordered sequence of
1547 relation names associated with the nodes of an explanation path, excluding the
1548 user node. Relation identifiers are mapped to relation names via `rid2relation`.
1549 The metric evaluates how many *distinct path patterns* are used per user.
1551 The score is computed per user as the ratio between the number of distinct path
1552 patterns and the minimum between the number of explanation paths and the maximum
1553 allowed path length, capped at 1.0. The final value is obtained by averaging
1554 across users.
1556 PPT is independent of the cutoff :math:`k`. The same value is reported for all
1557 configured values of `k`, consistently with the current hopwise implementation.
1559 Formally, let:
1560 - :math:`U` be the set of users,
1561 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1562 - :math:`\\Pi_u` be the set of distinct relational path patterns derived from
1563 :math:`P_u`,
1564 - :math:`L` be the maximum path length.
1566 The PPT metric is defined as:
1568 .. math::
1569 \mathrm{PPT}
1570 =
1571 \frac{1}{|U|}
1572 \sum_{u \in U}
1573 \min\!\left(
1574 \frac{|\Pi_u|}{\min\!\left(|P_u|,\; L\right)},
1575 1
1576 \right)
1578 Higher values indicate a greater variety of relational explanation patterns,
1579 reflecting richer and less repetitive explanation structures.
1580 """
1582 metric_need = ["data.max_path_length", "data.rid2relation"]
1584 def __init__(self, config):
1585 super().__init__(config)
1587 def calculate_metric(self, dataobject):
1588 paths = self.used_info(dataobject)
1589 max_path_pattern = dataobject.get("data.max_path_length")
1591 rid2r_name = {i: rel for i, rel in enumerate(dataobject.get("data.rid2relation"))}
1592 result = self.metric_info(paths, max_path_pattern, rid2r_name)
1593 metric_dict = self.topk_result("PPT", result)
1594 return metric_dict
1596 def metric_info(self, paths, max_path_pattern, rid2r_name):
1597 unique_path_pattern = dict()
1598 for user, _, _, path in paths:
1599 if user not in unique_path_pattern:
1600 unique_path_pattern[user] = [0, set()]
1601 path_pattern = [rid2r_name[path_tuple[0]] for path_tuple in path[1:]]
1602 unique_path_pattern[user][0] += 1
1603 unique_path_pattern[user][1].add("_".join(path_pattern))
1605 ppt = []
1606 for user in unique_path_pattern:
1607 n_paths = unique_path_pattern[user][0]
1608 n_path_patterns = len(unique_path_pattern[user][1])
1609 ppt.append(min(n_path_patterns / min(n_paths, max_path_pattern), 1.0))
1611 return np.array(ppt)
1614class LITD(PathQualityMetric):
1615 r"""LITD (Linked Interaction Type Diversity) is a path quality metric that measures
1616 the diversity of *types* of linking interactions used in explanation paths.
1618 Note:
1619 In this implementation, LITD evaluates how many distinct *linking interaction
1620 types* (e.g., item, entity, brand) are involved in a user's explanation paths.
1621 The linking interaction type is identified as the type of the second node in
1622 each explanation path.
1624 The metric is computed per user as the ratio between the number of distinct
1625 linking interaction types and the total number of explanation paths, and then
1626 averaged across users.
1628 LITD is independent of the cutoff :math:`k`. The same value is reported for all
1629 configured values of `k`, consistently with the current hopwise implementation.
1631 Formally, let:
1632 - :math:`U` be the set of users,
1633 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1634 - :math:`LT_u` be the set of distinct linking interaction types appearing
1635 as the second node of the paths in :math:`P_u`.
1637 The LITD metric is defined as:
1639 .. math::
1640 \mathrm{LITD}
1641 =
1642 \frac{1}{|U|}
1643 \sum_{u \in U}
1644 \frac{|LT_u|}{|P_u|}
1646 Higher values indicate that explanations rely on a wider variety of linking
1647 interaction types, reflecting more heterogeneous explanatory mechanisms.
1648 """
1650 def __init__(self, config):
1651 super().__init__(config)
1653 def calculate_metric(self, dataobject):
1654 paths = self.used_info(dataobject)
1655 result = self.metric_info(paths)
1656 metric_dict = self.topk_result("LITD", result)
1657 return metric_dict
1659 def metric_info(self, paths):
1660 unique_linking_interaction_types = dict()
1661 for user, _, _, path in paths:
1662 if user not in unique_linking_interaction_types:
1663 unique_linking_interaction_types[user] = [0, set()]
1664 linked_interaction_type = path[1][1]
1665 unique_linking_interaction_types[user][0] += 1
1666 unique_linking_interaction_types[user][1].add(linked_interaction_type)
1668 litd = []
1669 for user in unique_linking_interaction_types:
1670 n_paths = unique_linking_interaction_types[user][0]
1671 n_linked_interaction_types = len(unique_linking_interaction_types[user][1])
1672 litd.append(n_linked_interaction_types / n_paths)
1673 return np.array(litd)
1676class SETD(PathQualityMetric):
1677 r"""SETD (Shared Entity Type Diversity) is a path quality metric that measures
1678 the diversity of *types* of shared entities involved in explanation paths.
1680 Note:
1681 In this implementation, SETD evaluates how many distinct *shared entity types*
1682 (e.g., entity, brand, category) are used in a user's explanation paths.
1683 The shared entity type is identified as the type of the penultimate node in
1684 each explanation path.
1686 The metric is computed per user as the ratio between the number of distinct
1687 shared entity types and the total number of explanation paths, and then
1688 averaged across users.
1690 SETD is independent of the cutoff :math:`k`. The same value is reported for all
1691 configured values of `k`, consistently with the current hopwise implementation.
1693 Formally, let:
1694 - :math:`U` be the set of users,
1695 - :math:`P_u` be the set of explanation paths associated with user :math:`u`,
1696 - :math:`ST_u` be the set of distinct shared entity types appearing as the
1697 penultimate node of the paths in :math:`P_u`.
1699 The SETD metric is defined as:
1701 .. math::
1702 \mathrm{SETD}
1703 =
1704 \frac{1}{|U|}
1705 \sum_{u \in U}
1706 \frac{|ST_u|}{|P_u|}
1708 Higher values indicate that explanations rely on a wider variety of shared
1709 entity types, reflecting more heterogeneous explanatory structures.
1710 """
1712 def __init__(self, config):
1713 super().__init__(config)
1715 def calculate_metric(self, dataobject):
1716 paths = self.used_info(dataobject)
1717 result = self.metric_info(paths)
1718 metric_dict = self.topk_result("SETD", result)
1719 return metric_dict
1721 def metric_info(self, paths):
1722 unique_shared_entity_type = dict()
1723 for user, _, _, path in paths:
1724 if user not in unique_shared_entity_type:
1725 unique_shared_entity_type[user] = [0, set()]
1726 shared_entity_type = path[-2][1]
1727 unique_shared_entity_type[user][0] += 1
1728 unique_shared_entity_type[user][1].add(shared_entity_type)
1730 setd = []
1731 for user in unique_shared_entity_type:
1732 n_paths = unique_shared_entity_type[user][0]
1733 n_shared_entity_types = len(unique_shared_entity_type[user][1])
1734 setd.append(n_shared_entity_types / n_paths)
1735 return np.array(setd)