Skip to content

oversampleqa.distance

oversampleqa.distance

Distance metrics used within oversampleqa.

braycurtis_distance(x1, x2)

Compute Bray-Curtis distance between two vectors.

Often used in ecology and environmental science.

Source code in src/oversampleqa/extended_distances.py
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
def braycurtis_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating]
) -> float:
    """Compute Bray-Curtis distance between two vectors.

    Often used in ecology and environmental science.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    if np.any(x1 < 0) or np.any(x2 < 0):
        raise ValueError("Bray-Curtis distance requires non-negative inputs")

    numerator = np.sum(np.abs(x1 - x2))
    denominator = np.sum(np.abs(x1 + x2))

    if denominator == 0:
        # Sound only because the inputs are non-negative: the sum of absolute
        # values is then zero exactly when both vectors are all-zero, and the
        # distance between them really is zero. Allow a negative through and
        # the terms cancel instead -- d([-1, 0], [1, 0]) came out as 0.0, two
        # distinct points at distance zero, which is the identity-of-
        # indiscernibles violation check_metric_axioms exists to catch.
        return 0.0

    ratio: float = numerator / denominator
    return ratio

canberra_distance(x1, x2)

Compute Canberra distance between two vectors.

Canberra distance is a weighted version of Manhattan distance, useful when dealing with features of different scales.

Source code in src/oversampleqa/extended_distances.py
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
def canberra_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute Canberra distance between two vectors.

    Canberra distance is a weighted version of Manhattan distance,
    useful when dealing with features of different scales.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    numerator = np.abs(x1 - x2)
    denominator = np.abs(x1) + np.abs(x2)

    # Handle division by zero
    with np.errstate(divide="ignore", invalid="ignore"):
        ratio = np.where(denominator == 0, 0.0, numerator / denominator)

    return float(np.sum(ratio))

chebyshev_distance(x1, x2)

Compute Chebyshev (L-infinity) distance between two vectors.

This is the maximum absolute difference across all dimensions.

Source code in src/oversampleqa/extended_distances.py
63
64
65
66
67
68
69
70
71
72
73
74
75
def chebyshev_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating]
) -> float:
    """Compute Chebyshev (L-infinity) distance between two vectors.

    This is the maximum absolute difference across all dimensions.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    return float(np.max(np.abs(x1 - x2)))

correlation_distance(x1, x2)

Compute correlation distance between two vectors.

Correlation distance = 1 - Pearson correlation coefficient

Source code in src/oversampleqa/extended_distances.py
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
def correlation_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating]
) -> float:
    """Compute correlation distance between two vectors.

    Correlation distance = 1 - Pearson correlation coefficient
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    if len(x1) < 2:
        # Correlation needs at least two components to have any variance.
        raise ValueError(
            "correlation distance is undefined for vectors of length < 2: "
            "there is no variance to correlate."
        )

    with np.errstate(invalid="ignore", divide="ignore"):
        # A constant vector makes corrcoef divide by a zero standard deviation.
        # That is the case handled immediately below, so the warning is noise.
        corr_coef = np.corrcoef(x1, x2)[0, 1]

    if np.isnan(corr_coef):
        # A constant vector has zero variance, so the correlation is undefined.
        # This returned 0.0 -- "perfectly correlated" -- which made a constant
        # vector distance-zero from every other vector. METRIC_DOMAINS has
        # documented the case as undefined all along; the code disagreed.
        raise ValueError(
            "correlation distance is undefined when either vector is constant: "
            "zero variance leaves nothing to correlate. Drop constant features "
            "or rows, or use a metric defined on them such as 'euclidean'."
        )

    coefficient: float = corr_coef
    return float(np.clip(1.0 - coefficient, 0.0, 2.0))

energy_distance(x1, x2)

Compute energy distance between two 1D or 2D vectors.

The implementation follows the definition from energy statistics.

.. warning::

This is a sample-based metric, not a point metric. A 1-D input is reshaped to (len(x), 1) and treated as a set of scalar observations, not as one point in len(x)-dimensional feature space. It therefore does not measure the same kind of quantity as euclidean or hassanat, even though it is reachable through the same registry. Use it to compare two samples, not two points.

Source code in src/oversampleqa/extended_distances.py
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
def energy_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute energy distance between two 1D or 2D vectors.

    The implementation follows the definition from energy statistics.

    .. warning::

       This is a **sample-based** metric, not a point metric. A 1-D input is
       reshaped to ``(len(x), 1)`` and treated as a *set of scalar
       observations*, not as one point in ``len(x)``-dimensional feature
       space. It therefore does not measure the same kind of quantity as
       ``euclidean`` or ``hassanat``, even though it is reachable through the
       same registry. Use it to compare two samples, not two points.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)

    x1 = x1.reshape(len(x1), -1)
    x2 = x2.reshape(len(x2), -1)

    diff_cross = np.linalg.norm(x1[:, None, :] - x2[None, :, :], axis=-1)
    term_a = diff_cross.mean()

    term_b: float
    if len(x1) > 1:
        diff_x1 = np.linalg.norm(x1[:, None, :] - x1[None, :, :], axis=-1)
        term_b = float(diff_x1[np.triu_indices(len(x1), 1)].mean())
    else:
        term_b = 0.0

    term_c: float
    if len(x2) > 1:
        diff_x2 = np.linalg.norm(x2[:, None, :] - x2[None, :, :], axis=-1)
        term_c = float(diff_x2[np.triu_indices(len(x2), 1)].mean())
    else:
        term_c = 0.0

    return float(2.0 * term_a - term_b - term_c)

hamming_distance(x1, x2)

Compute Hamming distance between two vectors.

Counts the number of positions where elements differ. Useful for categorical or binary features.

Source code in src/oversampleqa/extended_distances.py
181
182
183
184
185
186
187
188
189
190
191
192
def hamming_distance(x1: NDArray[np.generic], x2: NDArray[np.generic]) -> float:
    """Compute Hamming distance between two vectors.

    Counts the number of positions where elements differ.
    Useful for categorical or binary features.
    """
    x1 = np.asarray(x1)
    x2 = np.asarray(x2)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    return float(np.sum(x1 != x2))

hellinger_distance(x1, x2)

Compute the Hellinger distance between two probability vectors.

The input vectors are normalized to sum to 1 and must contain non-negative values. The distance is bounded between 0 and 1.

Source code in src/oversampleqa/extended_distances.py
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
def hellinger_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute the Hellinger distance between two probability vectors.

    The input vectors are normalized to sum to ``1`` and must contain
    non-negative values. The distance is bounded between ``0`` and ``1``.
    """

    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    if np.any(x1 < 0) or np.any(x2 < 0):
        raise ValueError("Hellinger distance requires non-negative inputs")

    p = x1 / x1.sum() if x1.sum() != 0 else np.zeros_like(x1)
    q = x2 / x2.sum() if x2.sum() != 0 else np.zeros_like(x2)

    return float(np.linalg.norm(np.sqrt(p) - np.sqrt(q)) / np.sqrt(2.0))

jaccard_distance(x1, x2)

Compute Jaccard distance between two binary vectors.

Jaccard distance = 1 - Jaccard similarity where Jaccard similarity = :math:|intersection| / |union|

Source code in src/oversampleqa/extended_distances.py
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
def jaccard_distance(x1: NDArray[np.generic], x2: NDArray[np.generic]) -> float:
    """Compute Jaccard distance between two binary vectors.

    Jaccard distance = 1 - Jaccard similarity
    where Jaccard similarity = :math:`|intersection| / |union|`
    """
    x1_raw = np.asarray(x1)
    x2_raw = np.asarray(x2)
    x1_bool = x1_raw.astype(bool)
    x2_bool = x2_raw.astype(bool)
    if x1_bool.shape != x2_bool.shape:
        raise ValueError("Input vectors must have the same shape")
    if not _is_binary(x1_raw) or not _is_binary(x2_raw):
        # Without this, casting to bool made every non-zero identical:
        # d([1.0, 3.0], [7.0, 0.2]) was 0.0, two distinct points at distance
        # zero. `boolean` is the domain this metric declares in METRIC_DOMAINS.
        raise ValueError(
            "Jaccard distance requires binary inputs: values must be 0 or 1, "
            "or a boolean array. Casting other values to bool treats every "
            "non-zero as identical, so distinct points come out at distance "
            "zero. Binarise the features first, choosing the threshold "
            "deliberately."
        )

    intersection = np.sum(x1_bool & x2_bool)
    union = np.sum(x1_bool | x2_bool)

    if union == 0:
        return 0.0  # Both vectors are all zeros

    similarity: float = intersection / union
    return 1.0 - similarity

jensen_shannon_distance(x1, x2)

Compute the Jensen-Shannon distance between two probability vectors.

The Jensen-Shannon distance is the square root of the Jensen-Shannon divergence and is symmetric and bounded between 0 and sqrt(log(2)) when using natural logarithms.

Source code in src/oversampleqa/extended_distances.py
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
def jensen_shannon_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating]
) -> float:
    """Compute the Jensen-Shannon distance between two probability vectors.

    The Jensen-Shannon distance is the square root of the
    Jensen-Shannon divergence and is symmetric and bounded between ``0`` and
    ``sqrt(log(2))`` when using natural logarithms.
    """

    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    if np.any(x1 < 0) or np.any(x2 < 0):
        raise ValueError("Jensen-Shannon distance requires non-negative inputs")

    p = x1 / x1.sum() if x1.sum() != 0 else np.zeros_like(x1)
    q = x2 / x2.sum() if x2.sum() != 0 else np.zeros_like(x2)
    m = 0.5 * (p + q)

    def _kl_div(a: NDArray[np.floating], b: NDArray[np.floating]) -> float:
        with np.errstate(divide="ignore", invalid="ignore"):
            ratio = np.where(a == 0, 1.0, a / b)
            log_term = np.log(ratio)
        return float(np.sum(np.where(a == 0, 0.0, a * log_term)))

    js_div = 0.5 * _kl_div(p, m) + 0.5 * _kl_div(q, m)
    return float(np.sqrt(js_div))

mahalanobis_distance(x1, x2, cov_inv=None)

Compute Mahalanobis distance between two vectors.

Parameters

x1, x2 : np.ndarray Input vectors cov_inv : np.ndarray Inverse covariance matrix. Required, and must be symmetric positive semi-definite -- that is what makes the result a distance. It is not validated as such on every call, because an eigenvalue check per pair would cost more than the distance itself; a negative squared distance is caught instead, which is how a non-PSD matrix usually shows up.

Note the residual case: a matrix that is not PSD can still return 0
for two distinct points, and no per-pair check can detect that. If you
build ``cov_inv`` by any route other than inverting a sample
covariance, check it once with ``np.linalg.eigvalsh``.
Returns

float Mahalanobis distance

Raises

ValueError If cov_inv is omitted, or if it yields a negative squared distance.

Source code in src/oversampleqa/extended_distances.py
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
def mahalanobis_distance(
    x1: NDArray[np.floating],
    x2: NDArray[np.floating],
    cov_inv: NDArray[np.floating] | None = None,
) -> float:
    """Compute Mahalanobis distance between two vectors.

    Parameters
    ----------
    x1, x2 : np.ndarray
        Input vectors
    cov_inv : np.ndarray
        Inverse covariance matrix. Required, and must be symmetric positive
        semi-definite -- that is what makes the result a distance. It is not
        validated as such on every call, because an eigenvalue check per pair
        would cost more than the distance itself; a negative squared distance
        is caught instead, which is how a non-PSD matrix usually shows up.

        Note the residual case: a matrix that is not PSD can still return 0
        for two distinct points, and no per-pair check can detect that. If you
        build ``cov_inv`` by any route other than inverting a sample
        covariance, check it once with ``np.linalg.eigvalsh``.

    Returns
    -------
    float
        Mahalanobis distance

    Raises
    ------
    ValueError
        If ``cov_inv`` is omitted, or if it yields a negative squared distance.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    diff = x1 - x2

    if cov_inv is None:
        # Silently returning Euclidean relabels one metric as another. It was
        # doing exactly that in the advanced benchmark's default metric list,
        # where every "mahalanobis" row was a byte-identical copy of the
        # "euclidean" row -- double-weighting euclidean in the rankings and
        # making the pairwise correction treat one comparison as two.
        raise ValueError(
            "mahalanobis requires cov_inv: Mahalanobis distance with an "
            "identity covariance is Euclidean distance, so defaulting to it "
            "would report one metric under another's name. Estimate the "
            "inverse from the reference data, e.g. "
            "cov_inv=np.linalg.pinv(np.cov(X, rowvar=False)), and pass it "
            "through metric_kwargs."
        )

    squared = float(np.dot(diff, np.dot(cov_inv, diff)))
    if squared < 0.0:
        # A genuine inverse covariance is positive semi-definite, so this
        # quadratic form cannot be negative. When it is, np.sqrt returns nan
        # with nothing but a bare "invalid value encountered in sqrt" to say
        # why -- a warning users routinely filter, pointing at a line inside
        # this library rather than at the matrix they passed.
        #
        # A near-singular inverse can produce a tiny negative through rounding
        # alone, which is noise rather than an error, so that is clamped.
        # Anything larger means cov_inv is not an inverse covariance.
        tolerance = 1e-12 * max(1.0, float(np.dot(diff, diff)))
        if squared < -tolerance:
            raise ValueError(
                "mahalanobis requires a positive semi-definite cov_inv: the "
                f"squared distance came out negative ({squared:.6g}), which "
                "np.sqrt reports as nan. Passing the covariance itself rather "
                "than its inverse, or inverting a covariance estimated from "
                "fewer samples than features, both produce a matrix that is "
                "not. Estimate it with "
                "cov_inv=np.linalg.pinv(np.cov(X, rowvar=False))."
            )
        squared = 0.0

    return float(np.sqrt(squared))

minkowski_distance(x1, x2, p=3.0)

Compute Minkowski distance between two vectors.

Parameters

x1, x2 : np.ndarray Input vectors of same shape p : float, default=3.0 Order of the norm (p >= 1). np.inf is accepted and gives the Chebyshev distance, which is the limit as p grows.

Returns

float Minkowski distance

Raises

ValueError If the shapes differ, or p < 1.

Source code in src/oversampleqa/extended_distances.py
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
def minkowski_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating], p: float = 3.0
) -> float:
    """Compute Minkowski distance between two vectors.

    Parameters
    ----------
    x1, x2 : np.ndarray
        Input vectors of same shape
    p : float, default=3.0
        Order of the norm (``p >= 1``). ``np.inf`` is accepted and gives the
        Chebyshev distance, which is the limit as p grows.

    Returns
    -------
    float
        Minkowski distance

    Raises
    ------
    ValueError
        If the shapes differ, or ``p < 1``.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    if p < 1:
        raise ValueError("p must be >= 1")

    diff = np.abs(x1 - x2)
    largest = float(diff.max()) if diff.size else 0.0
    if largest == 0.0:
        return 0.0

    if np.isinf(p):
        # The p -> infinity limit is the Chebyshev distance. The general
        # formula cannot produce it: every |d| > 1 raised to inf is inf, the
        # sum is inf, and inf ** (1 / inf) is inf ** 0, which is 1.0. So
        # `p=inf` returned 1.0 for any input at all, regardless of the data.
        return largest

    # The largest term is factored out before exponentiating. Computed
    # directly, `diff ** p` overflows at moderate p -- p=1000 gives inf, as it
    # does in scipy -- when the answer is simply the largest term. Scaling
    # every term into [0, 1] first makes the sum well behaved, and the result
    # is identical for the p values that never overflowed.
    scaled = diff / largest
    return float(largest * np.sum(scaled**p) ** (1 / p))

wasserstein_1d_distance(x1, x2)

Compute the 1D Wasserstein distance between two empirical distributions.

.. warning::

This is a sample-based metric, not a point metric. The input vector is flattened and treated as a set of scalar observations drawn from a distribution, not as one point in feature space. It therefore does not measure the same kind of quantity as euclidean or hassanat, even though it is reachable through the same registry. Use it to compare two samples, not two points.

Parameters:

Name Type Description Default
x1 NDArray[floating]

Samples from distribution 1.

required
x2 NDArray[floating]

Samples from distribution 2.

required

Returns:

Type Description
float

Wasserstein distance.

Source code in src/oversampleqa/extended_distances.py
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
def wasserstein_1d_distance(
    x1: NDArray[np.floating], x2: NDArray[np.floating]
) -> float:
    """Compute the 1D Wasserstein distance between two empirical distributions.

    .. warning::

       This is a **sample-based** metric, not a point metric. The input vector
       is flattened and treated as a *set of scalar observations* drawn from a
       distribution, not as one point in feature space. It therefore does not
       measure the same kind of quantity as ``euclidean`` or ``hassanat``,
       even though it is reachable through the same registry. Use it to
       compare two samples, not two points.

    Args:
        x1: Samples from distribution 1.
        x2: Samples from distribution 2.

    Returns:
        Wasserstein distance.
    """
    x1 = np.sort(np.asarray(x1, dtype=float).ravel())
    x2 = np.sort(np.asarray(x2, dtype=float).ravel())

    n = len(x1)
    m = len(x2)
    if n == 0 or m == 0:
        return 0.0

    i = j = 0
    cdf1 = cdf2 = 0.0
    last_x = min(x1[0], x2[0])
    dist = 0.0

    # W1 = integral of |F1(t) - F2(t)| dt. Between consecutive sorted points the
    # two CDFs are flat, so each interval [last_x, x) contributes
    # |F1 - F2| * (x - last_x) using the CDF values that hold *across* it --
    # that is, the values before the jump at x.
    #
    # The previous version advanced the CDF first and then added, crediting each
    # interval with the value from after its right-hand jump. On [0, 1] vs
    # [0, 3] that returned 0.5 where the true W1 is 1.0.
    while i < n and j < m:
        x = x1[i] if x1[i] <= x2[j] else x2[j]
        dist += abs(cdf1 - cdf2) * (x - last_x)
        last_x = x
        # Advance every tie at this position before moving on.
        while i < n and x1[i] == x:
            i += 1
            cdf1 = i / n
        while j < m and x2[j] == x:
            j += 1
            cdf2 = j / m

    while i < n:
        x = x1[i]
        dist += abs(cdf1 - cdf2) * (x - last_x)
        last_x = x
        i += 1
        cdf1 = i / n

    while j < m:
        x = x2[j]
        dist += abs(cdf1 - cdf2) * (x - last_x)
        last_x = x
        j += 1
        cdf2 = j / m

    return float(dist)

hassanat_distance(x1, x2)

Compute the Hassanat distance between two vectors.

For each dimension :math:i, with :math:m = \min(a_i, b_i) and :math:M = \max(a_i, b_i):

.. math::

D(a_i, b_i) = \begin{cases} 1 - \dfrac{1 + m}{1 + M} & m \ge 0 \[2ex] 1 - \dfrac{1 + m + |m|}{1 + M + |m|} & m < 0 \end{cases}

and :math:HD(a, b) = \sum_i D(a_i, b_i).

Every per-dimension term lies in :math:[0, 1), which is what makes the metric invariant to feature scale and robust to outliers: no single dimension can contribute more than 1 regardless of its magnitude.

Parameters

x1, x2 : NDArray[np.floating] Input vectors of identical shape.

Returns

float Hassanat distance, in [0, n_features).

Raises

ValueError If the two vectors do not have the same shape.

References

Hassanat, A. B. (2014). Dimensionality invariant similarity measure. Journal of American Science, 10(8).

Source code in src/oversampleqa/distance.py
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
def hassanat_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    r"""Compute the Hassanat distance between two vectors.

    For each dimension :math:`i`, with :math:`m = \min(a_i, b_i)` and
    :math:`M = \max(a_i, b_i)`:

    .. math::

       D(a_i, b_i) = \begin{cases}
         1 - \dfrac{1 + m}{1 + M} & m \ge 0 \\[2ex]
         1 - \dfrac{1 + m + |m|}{1 + M + |m|} & m < 0
       \end{cases}

    and :math:`HD(a, b) = \sum_i D(a_i, b_i)`.

    Every per-dimension term lies in :math:`[0, 1)`, which is what makes the
    metric invariant to feature scale and robust to outliers: no single
    dimension can contribute more than 1 regardless of its magnitude.

    Parameters
    ----------
    x1, x2 : NDArray[np.floating]
        Input vectors of identical shape.

    Returns
    -------
    float
        Hassanat distance, in ``[0, n_features)``.

    Raises
    ------
    ValueError
        If the two vectors do not have the same shape.

    References
    ----------
    Hassanat, A. B. (2014). Dimensionality invariant similarity measure.
    *Journal of American Science*, 10(8).
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")

    mn = np.minimum(x1, x2)
    mx = np.maximum(x1, x2)
    # Adding |min| on the negative branch shifts both terms up so the ratio
    # stays in (0, 1]. The denominator is 1 + mx + shift; since mx >= mn and
    # shift = max(-mn, 0), we have mx + shift >= mn + shift >= 0, so the
    # denominator is >= 1 and can never vanish. No division guard is needed.
    shift = np.where(mn < 0.0, -mn, 0.0)
    return float(np.sum(1.0 - (1.0 + mn + shift) / (1.0 + mx + shift)))

euclidean_distance(x1, x2)

Compute Euclidean distance between two vectors.

Parameters:

Name Type Description Default
x1 NDArray[floating]

First vector.

required
x2 NDArray[floating]

Second vector.

required

Returns:

Type Description
float

Euclidean distance.

Source code in src/oversampleqa/distance.py
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
def euclidean_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute Euclidean distance between two vectors.

    Args:
        x1: First vector.
        x2: Second vector.

    Returns:
        Euclidean distance.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    return float(np.linalg.norm(x1 - x2))

manhattan_distance(x1, x2)

Compute Manhattan distance between two vectors.

Parameters:

Name Type Description Default
x1 NDArray[floating]

First vector.

required
x2 NDArray[floating]

Second vector.

required

Returns:

Type Description
float

Manhattan distance.

Source code in src/oversampleqa/distance.py
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
def manhattan_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute Manhattan distance between two vectors.

    Args:
        x1: First vector.
        x2: Second vector.

    Returns:
        Manhattan distance.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    return float(np.sum(np.abs(x1 - x2)))

cosine_distance(x1, x2)

Compute Cosine distance between two vectors.

Parameters:

Name Type Description Default
x1 NDArray[floating]

First vector.

required
x2 NDArray[floating]

Second vector.

required

Returns:

Type Description
float

Cosine distance.

Source code in src/oversampleqa/distance.py
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
def cosine_distance(x1: NDArray[np.floating], x2: NDArray[np.floating]) -> float:
    """Compute Cosine distance between two vectors.

    Args:
        x1: First vector.
        x2: Second vector.

    Returns:
        Cosine distance.
    """
    x1 = np.asarray(x1, dtype=float)
    x2 = np.asarray(x2, dtype=float)
    if x1.shape != x2.shape:
        raise ValueError("Input vectors must have the same shape")
    dot = np.dot(x1, x2)
    denom = np.linalg.norm(x1) * np.linalg.norm(x2)
    if denom == 0:
        # A zero vector has no direction, so the angle to it is undefined.
        # Returning 0.0 said "identical direction", which made the zero vector
        # distance-zero from *every* vector -- the identity of indiscernibles
        # broken in the same way the original hassanat defect broke it. The
        # randomised axiom check never draws an exact zero vector, so it never
        # saw this.
        raise ValueError(
            "cosine distance is undefined when either vector is zero: a zero "
            "vector has no direction. Drop all-zero rows, or use a metric "
            "defined on them such as 'euclidean' or 'hassanat'."
        )
    # Clamp: floating point put d(x, x) at -2.22e-16 for a constant vector, a
    # negative distance feeding a nearest-neighbour comparison.
    return float(np.clip(1.0 - dot / denom, 0.0, 2.0))

resolve_metric(name)

Return a plugin metric callable, or None for a built-in.

Registering a metric plugin used to accomplish nothing beyond making it retrievable from the registry: distance_matrix and every validator that funnels through it consulted only the built-in table, so a plugin metric was rejected as unsupported by the exact functions it exists to be used by.

Resolution happens per call rather than at import, because plugins register at runtime -- often from discover_entry_points -- and the built-in table is bound when this module is imported.

Parameters:

Name Type Description Default
name str

Metric identifier.

required

Returns:

Type Description
MetricFunc | None

A callable for a registered plugin metric, or None when the name is

MetricFunc | None

a built-in and the default registry already covers it.

Raises:

Type Description
ValueError

If the name is neither a built-in nor a registered plugin.

Source code in src/oversampleqa/distance.py
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
def resolve_metric(name: str) -> MetricFunc | None:
    """Return a plugin metric callable, or ``None`` for a built-in.

    Registering a metric plugin used to accomplish nothing beyond making it
    retrievable from the registry: ``distance_matrix`` and every validator that
    funnels through it consulted only the built-in table, so a plugin metric was
    rejected as unsupported by the exact functions it exists to be used by.

    Resolution happens per call rather than at import, because plugins register
    at runtime -- often from ``discover_entry_points`` -- and the built-in table
    is bound when this module is imported.

    Args:
        name: Metric identifier.

    Returns:
        A callable for a registered plugin metric, or ``None`` when the name is
        a built-in and the default registry already covers it.

    Raises:
        ValueError: If the name is neither a built-in nor a registered plugin.
    """
    if name in _METRICS:
        return None

    # Local import: plugin_system reads _METRICS from this module, so importing
    # it at module scope would be a cycle.
    from .plugin_system import plugin_manager

    try:
        registered = plugin_manager.get_metric(name)
    except KeyError:
        known = ", ".join(sorted(_METRICS))
        raise ValueError(
            f"Unsupported metric {name!r}. Built-in metrics: {known}. "
            "If this is a plugin, register it first -- "
            "plugin_manager.discover_entry_points() for an installed package, "
            "or plugin_manager.register_metric(...) directly."
        ) from None
    return registered() if isinstance(registered, type) else registered

distance_matrix(X1, X2, metric='hassanat', *, batch_size='auto', cache=None, **metric_kwargs)

Compute pairwise distance matrix using the given metric.

Parameters

X1, X2 : ndarray Input matrices containing observations. metric : str, default="hassanat" Identifier of the distance metric to use. batch_size : int or {"auto", "stream"}, default="auto" Controls batching strategy. "auto" selects a batch size that fits memory_limit_gb of :class:OptimizedDistanceMatrix. "stream" forces row-wise streaming when memory is constrained. cache : ValidationCache, optional Opt-in cache. Caching is off by default: nothing is written to disk and no directory is created unless you supply one. Worth it for expensive metrics such as hassanat; a net loss for euclidean, where hashing the inputs costs more than recomputing the result. **metric_kwargs : Additional keyword arguments are forwarded to the metric function. This enables configuration of metrics that require extra parameters, such as the inverse covariance matrix for Mahalanobis distance.

Returns

ndarray Distance matrix. When cache is supplied the array is read-only; call .copy() before modifying it.

Source code in src/oversampleqa/distance.py
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
def distance_matrix(
    X1: NDArray[np.floating],
    X2: NDArray[np.floating],
    metric: str = "hassanat",
    *,
    batch_size: int | str = "auto",
    cache: ValidationCache | None = None,
    **metric_kwargs: Any,
) -> NDArray[np.floating]:
    """Compute pairwise distance matrix using the given metric.

    Parameters
    ----------
    X1, X2 : ndarray
        Input matrices containing observations.
    metric : str, default="hassanat"
        Identifier of the distance metric to use.
    batch_size : int or {"auto", "stream"}, default="auto"
        Controls batching strategy. ``"auto"`` selects a batch size that fits
        ``memory_limit_gb`` of :class:`OptimizedDistanceMatrix`. ``"stream"``
        forces row-wise streaming when memory is constrained.
    cache : ValidationCache, optional
        Opt-in cache. Caching is off by default: nothing is written to disk and
        no directory is created unless you supply one. Worth it for expensive
        metrics such as ``hassanat``; a net loss for ``euclidean``, where
        hashing the inputs costs more than recomputing the result.
    **metric_kwargs :
        Additional keyword arguments are forwarded to the metric function. This
        enables configuration of metrics that require extra parameters, such as
        the inverse covariance matrix for Mahalanobis distance.

    Returns
    -------
    ndarray
        Distance matrix. When ``cache`` is supplied the array is **read-only**;
        call ``.copy()`` before modifying it.
    """
    plugin = resolve_metric(metric)
    registry = _METRICS if plugin is None else {**_METRICS, metric: plugin}
    metric_kwargs = metric_kwargs or {}
    optimizer = (
        _OPTIMIZER
        if cache is None and plugin is None
        else OptimizedDistanceMatrix(metric_registry=registry, cache=cache)
    )
    return optimizer.compute_distance_matrix(
        X1,
        X2,
        metric=metric,
        batch_size=batch_size,
        **metric_kwargs,
    )