Skip to content

Metrics

Metric computation for run evaluation.

regression_metrics and classification_metrics mirror the metric set the OpenML server stores (Weka's Evaluation vocabulary: prior errors, relative errors, kappa, KB relative information, ...). Golden values are pinned by tests/runs/test_metrics.py.

classification_metrics(y_true, y_pred, conf, class_names, y_train_labels, cost_matrix=None)

Source code in src/runs/metrics.py
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
def classification_metrics(
    y_true: np.ndarray,
    y_pred: np.ndarray,
    conf: np.ndarray,
    class_names: list[str],
    y_train_labels: Iterable,
    cost_matrix: Optional[np.ndarray] = None,
) -> dict:
    y_true = _encode_labels(y_true, class_names)
    y_pred = _encode_labels(y_pred, class_names)
    conf = np.asarray(conf, dtype=float)
    y_train_arr = _encode_labels(y_train_labels, class_names)
    n = len(y_true)
    num_classes = len(class_names)
    classes = list(range(num_classes))

    cm = confusion_matrix(y_true, y_pred, labels=classes)

    # Probability-vector MAE/RMSE (Weka's definition for classification).
    # Weka's Evaluation.meanAbsoluteError() divides by N * num_classes (the
    # per-class differences are summed inside, then the total is divided by
    # m_WithClass which is N — but the accumulation runs over classes too, so
    # the effective denominator is N*C). Confirmed against the OpenML server
    # for run 1 / dataset 9 (autos): port gave 0.8895, server had 0.1271,
    # ratio is exactly 7 = num_classes.
    onehot = np.eye(num_classes)[y_true]
    mae = float(np.abs(conf - onehot).mean())
    rmse = float(np.sqrt(np.square(conf - onehot).mean()))

    prior = class_ratios(y_train_arr, num_classes)
    mae_prior = float(np.abs(onehot - prior).mean())
    rmse_prior = float(np.sqrt(np.square(onehot - prior).mean()))
    h_prior = float(entropy(prior, base=2)) if prior.sum() > 0 else 0.0

    p, r, f, _ = precision_recall_fscore_support(
        y_true, y_pred, labels=classes, average=None, zero_division=0
    )
    instances_per_class = cm.sum(axis=1)
    total_inst = int(instances_per_class.sum())
    weights = (
        instances_per_class / total_inst if total_inst > 0 else np.zeros(num_classes)
    )

    try:
        auroc_per_class = roc_auc_score(
            np.eye(num_classes)[y_true],
            conf,
            average=None,
            multi_class="ovr",
            labels=classes,
        )
        auroc_per_class = [float(x) for x in auroc_per_class]
        auroc_global = (
            float(np.average(auroc_per_class, weights=weights))
            if weights.sum() > 0
            else None
        )
    except (ValueError, IndexError):
        auroc_per_class = [None] * num_classes
        auroc_global = None

    result: dict = {
        "mean_absolute_error": mae,
        "mean_prior_absolute_error": mae_prior,
        "root_mean_squared_error": rmse,
        "root_mean_prior_squared_error": rmse_prior,
        "relative_absolute_error": (mae / mae_prior) if mae_prior > 0 else 0.0,
        "root_relative_squared_error": (rmse / rmse_prior) if rmse_prior > 0 else 0.0,
        "prior_entropy": h_prior,
        "kb_relative_information_score": kb_relative_information(y_true, conf, prior),
        "predictive_accuracy": float(accuracy_score(y_true, y_pred)),
        "kappa": float(cohen_kappa_score(y_true, y_pred)),
        "precision": (
            float(np.average(p, weights=weights)) if weights.sum() > 0 else 0.0
        ),
        "weighted_recall": (
            float(np.average(r, weights=weights)) if weights.sum() > 0 else 0.0
        ),
        "unweighted_recall": float(r.mean()),
        "f_measure": (
            float(np.average(f, weights=weights)) if weights.sum() > 0 else 0.0
        ),
        "number_of_instances": float(n),
        "confusion_matrix": cm.tolist(),
        "_per_class": {
            "precision": p.tolist(),
            "recall": r.tolist(),
            "f_measure": f.tolist(),
            "auroc": auroc_per_class,
            "instances_per_class": instances_per_class.tolist(),
        },
    }
    if auroc_global is not None:
        result["area_under_roc_curve"] = auroc_global

    if cost_matrix is not None:
        cm_cost = np.asarray(cost_matrix, dtype=float)
        total_cost = float((cm * cm_cost).sum())
        result["total_cost"] = total_cost
        result["average_cost"] = total_cost / n if n > 0 else 0.0

    return result

kb_relative_information(y_true, conf, prior)

Source code in src/runs/metrics.py
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
def kb_relative_information(
    y_true: np.ndarray, conf: np.ndarray, prior: np.ndarray
) -> float:
    eps = 1e-12
    prior = np.asarray(prior, dtype=float)
    if not (prior > 0).any():
        return 0.0
    h_prior = float(entropy(prior, base=2))
    if h_prior == 0.0:
        return 0.0

    y_true = np.asarray(y_true, dtype=int)
    conf = np.asarray(conf, dtype=float)

    p_true_pred = conf[np.arange(len(y_true)), y_true]
    p_true_pred = np.clip(p_true_pred, eps, 1.0)
    p_true_prior = prior[y_true]
    p_true_prior = np.clip(p_true_prior, eps, 1.0)

    info_pred = -np.log2(p_true_pred)
    info_prior = -np.log2(p_true_prior)
    kb = float((info_prior - info_pred).mean())
    return kb / h_prior

regression_metrics(y_true, y_pred, y_train)

Source code in src/runs/metrics.py
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
def regression_metrics(
    y_true: np.ndarray, y_pred: np.ndarray, y_train: np.ndarray
) -> dict[str, float]:

    y_true = np.asarray(y_true, dtype=float)
    y_pred = np.asarray(y_pred, dtype=float)
    y_train = np.asarray(y_train, dtype=float)
    n = len(y_true)
    prior = float(np.mean(y_train))

    mae = float(mean_absolute_error(y_true, y_pred))
    rmse = float(root_mean_squared_error(y_true, y_pred))
    mae_prior = float(mean_absolute_error(y_true, np.full(n, prior)))
    rmse_prior = float(root_mean_squared_error(y_true, np.full(n, prior)))

    return {
        "mean_absolute_error": mae,
        "mean_prior_absolute_error": mae_prior,
        "number_of_instances": float(n),
        "root_mean_squared_error": rmse,
        "root_mean_prior_squared_error": rmse_prior,
        "relative_absolute_error": (mae / mae_prior) if mae_prior > 0 else 0.0,
        "root_relative_squared_error": (rmse / rmse_prior) if rmse_prior > 0 else 0.0,
    }