Skip to content

Stress Metrics

StressMetrics

smds.stress.stress_metrics.StressMetrics

Bases: Enum

Source code in smds/stress/stress_metrics.py
4
5
6
7
8
9
class StressMetrics(Enum):
    SCALE_NORMALIZED_STRESS = "scale_normalized_stress"
    NON_METRIC_STRESS = "non_metric_stress"
    SHEPARD_GOODNESS_SCORE = "shepard_goodness_score"
    NORMALIZED_STRESS = "normalized_stress"
    NORMALIZED_KL_DIVERGENCE = "normalized_kl_divergence"

Functions

scale_normalized_stress

smds.stress.scale_normalized_stress.scale_normalized_stress

scale_normalized_stress(
    d_true: NDArray[float64], d_pred: NDArray[float64]
) -> float

Compute the scale-normalized stress between true dissimilarities and embedding distances.

Parameters:

Name Type Description Default
d_true array-like of shape (n_pairs,)

The target dissimilarities (D_high/D_ideal). Expected to be a 1D array of flattened pairwise distances.

required
d_pred array-like of shape (n_pairs,)

The embedding distances (D_low/D_pred). Expected to be a 1D array of flattened pairwise distances.

required

Returns:

Name Type Description
stress float

The scale-normalized stress value.

References
  • Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
Source code in smds/stress/scale_normalized_stress.py
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
@validate_params(  # type: ignore[misc]
    {
        "d_true": ["array-like"],
        "d_pred": ["array-like"],
    },
    prefer_skip_nested_validation=True,
)
def scale_normalized_stress(d_true: NDArray[np.float64], d_pred: NDArray[np.float64]) -> float:
    """
    Compute the scale-normalized stress between true dissimilarities and embedding distances.

    Parameters
    ----------
    d_true : array-like of shape (n_pairs,)
        The target dissimilarities (D_high/D_ideal). Expected to be a 1D array
        of flattened pairwise distances.

    d_pred : array-like of shape (n_pairs,)
        The embedding distances (D_low/D_pred). Expected to be a 1D array
        of flattened pairwise distances.

    Returns
    -------
    stress : float
        The scale-normalized stress value.

    References
    ----------
    - Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not
    Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
    """
    d_true = check_array(d_true, ensure_2d=False, dtype=np.float64)
    d_pred = check_array(d_pred, ensure_2d=False, dtype=np.float64)
    check_consistent_length(d_true, d_pred)

    denominator_alpha = np.sum(d_pred**2)

    if denominator_alpha == 0:
        return np.inf

    alpha = np.sum(d_true * d_pred) / denominator_alpha

    residuals = d_true - (alpha * d_pred)
    denominator_d_true = np.sum(d_true**2)

    if denominator_d_true == 0:
        return np.inf

    result: float = float(np.sqrt(np.sum(residuals**2) / denominator_d_true))
    return result

normalized_stress

smds.stress.normalized_stress.normalized_stress

normalized_stress(
    d_true: NDArray[float64], d_pred: NDArray[float64]
) -> float

Compute the Normalized Stress between ideal and recovered geometries.

This metric quantifies the preservation of pairwise distances by comparing the squared differences relative to the magnitude of the ideal distances.

Parameters:

Name Type Description Default
d_true array-like of shape (n_pairs,)

The ideal/target distance matrix (D_high). Can be a flattened array of pairwise distances.

required
d_pred array-like of shape (n_pairs,)

The recovered/embedding distance matrix (D_low). Can be a flattened array of pairwise distances.

required

Returns:

Name Type Description
stress float

The calculated normalized stress value.

Notes

The formula implemented is:

.. math::

S = \frac{\sum (d_{pred} - d_{true})^2}{\sum d_{true}^2}
References
  • Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
Source code in smds/stress/normalized_stress.py
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
@validate_params(  # type: ignore[misc]
    {
        "d_true": ["array-like"],
        "d_pred": ["array-like"],
    },
    prefer_skip_nested_validation=True,
)
def normalized_stress(d_true: NDArray[np.float64], d_pred: NDArray[np.float64]) -> float:
    """
    Compute the Normalized Stress between ideal and recovered geometries.

    This metric quantifies the preservation of pairwise distances by comparing
    the squared differences relative to the magnitude of the ideal distances.

    Parameters
    ----------
    d_true : array-like of shape (n_pairs,)
        The ideal/target distance matrix (D_high).
        Can be a flattened array of pairwise distances.

    d_pred : array-like of shape (n_pairs,)
        The recovered/embedding distance matrix (D_low).
        Can be a flattened array of pairwise distances.

    Returns
    -------
    stress : float
        The calculated normalized stress value.

    Notes
    -----
    The formula implemented is:

    .. math::

        S = \\frac{\\sum (d_{pred} - d_{true})^2}{\\sum d_{true}^2}


    References
    ----------
    - Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not
    Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
    """
    d_true = check_array(d_true, ensure_2d=False, dtype=np.float64)
    d_pred = check_array(d_pred, ensure_2d=False, dtype=np.float64)
    check_consistent_length(d_true, d_pred)

    numerator = np.sum((d_pred - d_true) ** 2)
    denominator = np.sum(d_true**2)

    if denominator == 0:
        return np.inf
    result: float = float(np.sqrt(numerator / denominator))

    return result

non_metric_stress

smds.stress.non_metric_stress.non_metric_stress

non_metric_stress(
    d_true: NDArray[float64], d_pred: NDArray[float64]
) -> float

Compute the non-metric stress between true dissimilarities and embedding distances.

Parameters:

Name Type Description Default
d_true array-like of shape (n_pairs,)

The target dissimilarities (D_high/D_ideal). Expected to be a 1D array of flattened pairwise distances.

required
d_pred array-like of shape (n_pairs,)

The embedding distances (D_low/D_pred). Expected to be a 1D array of flattened pairwise distances.

required

Returns:

Name Type Description
stress float

The non-metric stress value.

References
  • Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
Source code in smds/stress/non_metric_stress.py
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
@validate_params(  # type: ignore[misc]
    {
        "d_true": ["array-like"],
        "d_pred": ["array-like"],
    },
    prefer_skip_nested_validation=True,
)
def non_metric_stress(d_true: NDArray[np.float64], d_pred: NDArray[np.float64]) -> float:
    """
    Compute the non-metric stress between true dissimilarities and embedding distances.

    Parameters
    ----------
    d_true : array-like of shape (n_pairs,)
        The target dissimilarities (D_high/D_ideal). Expected to be a 1D array
        of flattened pairwise distances.

    d_pred : array-like of shape (n_pairs,)
        The embedding distances (D_low/D_pred). Expected to be a 1D array
        of flattened pairwise distances.

    Returns
    -------
    stress : float
        The non-metric stress value.

    References
    ----------
    - Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not
    Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
    """
    d_true = check_array(d_true, ensure_2d=False, dtype=np.float64)
    d_pred = check_array(d_pred, ensure_2d=False, dtype=np.float64)
    check_consistent_length(d_true, d_pred)

    ir = IsotonicRegression(increasing=True)
    d_hat = ir.fit_transform(d_true, d_pred)

    numerator = np.sum((d_hat - d_pred) ** 2)
    denominator = np.sum(d_pred**2)

    if denominator == 0:
        return np.inf

    result: float = float(numerator / denominator)
    return result

shepard_goodness_stress

smds.stress.shepard_goodness_score.shepard_goodness_stress

shepard_goodness_stress(
    d_true: NDArray[float64], d_pred: NDArray[float64]
) -> float

Compute the Shepard Goodness Score (Spearman's Rho) on pairwise distances.

Parameters:

Name Type Description Default
d_true array-like of shape (n_pairs,)

The target dissimilarities (D_high/D_ideal).

required
d_pred array-like of shape (n_pairs,)

The embedding distances (D_low/D_pred).

required

Returns:

Name Type Description
score float

Spearman's rank correlation coefficient (rho).

References
  • Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
Source code in smds/stress/shepard_goodness_score.py
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
@validate_params(  # type: ignore[misc]
    {
        "d_true": ["array-like"],
        "d_pred": ["array-like"],
    },
    prefer_skip_nested_validation=True,
)
def shepard_goodness_stress(d_true: NDArray[np.float64], d_pred: NDArray[np.float64]) -> float:
    """
    Compute the Shepard Goodness Score (Spearman's Rho) on pairwise distances.

    Parameters
    ----------
    d_true : array-like of shape (n_pairs,)
        The target dissimilarities (D_high/D_ideal).

    d_pred : array-like of shape (n_pairs,)
        The embedding distances (D_low/D_pred).

    Returns
    -------
    score : float
        Spearman's rank correlation coefficient (rho).

    References
    ----------
    - Smelser, K., Miller, J., & Kobourov, S. (2024). "Normalized Stress is Not
    Normalized: How to Interpret Stress Correctly". arXiv preprint arXiv:2408.07724.
    """
    d_true = check_array(d_true, ensure_2d=False, dtype=np.float64)
    d_pred = check_array(d_pred, ensure_2d=False, dtype=np.float64)
    check_consistent_length(d_true, d_pred)

    correlation = spearmanr(d_true, d_pred)

    result: float = float(correlation[0])
    return result

kl_divergence_stress

smds.stress.kl_divergence.kl_divergence_stress

kl_divergence_stress(
    d_true: NDArray[float64],
    d_pred: NDArray[float64],
    sigma: float = 1.0,
) -> float

Compute the Kullback-Leibler (KL) Divergence using Gaussian kernels for both P and Q.

This metric measures the information loss when approximating the high-dimensional structure (P) with the low-dimensional structure (Q). Unlike standard t-SNE which uses a Student-t distribution for Q, this implementation uses a Gaussian kernel for both, corresponding to the metric denoted as :math:KL_G in Smelser et al. (2025).

Parameters:

Name Type Description Default
d_true array-like of shape (n_samples, n_samples)

The target distance matrix (Ground Truth/Ideal). Must be a square, symmetric matrix of pairwise distances.

required
d_pred array-like of shape (n_samples, n_samples)

The recovered distance matrix (Embedding). Must be a square, symmetric matrix of pairwise distances.

required
sigma float

The standard deviation (width) of the Gaussian kernel.

1.0

Returns:

Name Type Description
kl_div float

The Kullback-Leibler divergence sum(P * log(P / Q)).

References

Smelser, K., Gunaratne, K., Miller, J., & Kobourov, S. (2025). "How Scale Breaks 'Normalized Stress' and KL Divergence: Rethinking Quality Metrics". arXiv preprint arXiv:2510.08660.

Source code in smds/stress/kl_divergence.py
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
@validate_params(  # type: ignore[misc]
    {
        "d_true": ["array-like"],
        "d_pred": ["array-like"],
    },
    prefer_skip_nested_validation=True,
)
def kl_divergence_stress(d_true: NDArray[np.float64], d_pred: NDArray[np.float64], sigma: float = 1.0) -> float:
    """
    Compute the Kullback-Leibler (KL) Divergence using Gaussian kernels for both P and Q.

    This metric measures the information loss when approximating the high-dimensional
    structure (P) with the low-dimensional structure (Q). Unlike standard t-SNE which
    uses a Student-t distribution for Q, this implementation uses a Gaussian kernel
    for both, corresponding to the metric denoted as :math:`KL_G` in Smelser et al. (2025).

    Parameters
    ----------
    d_true : array-like of shape (n_samples, n_samples)
        The target distance matrix (Ground Truth/Ideal).
        Must be a square, symmetric matrix of pairwise distances.

    d_pred : array-like of shape (n_samples, n_samples)
        The recovered distance matrix (Embedding).
        Must be a square, symmetric matrix of pairwise distances.

    sigma : float, default=1.0
        The standard deviation (width) of the Gaussian kernel.

    Returns
    -------
    kl_div : float
        The Kullback-Leibler divergence sum(P * log(P / Q)).

    References
    ----------
    Smelser, K., Gunaratne, K., Miller, J., & Kobourov, S. (2025).
    "How Scale Breaks 'Normalized Stress' and KL Divergence: Rethinking Quality Metrics".
    arXiv preprint arXiv:2510.08660.
    """
    d_true = check_array(d_true, ensure_2d=True, dtype=np.float64)
    d_pred = check_array(d_pred, ensure_2d=True, dtype=np.float64)
    check_consistent_length(d_true, d_pred)

    if d_true.shape[0] != d_true.shape[1]:
        raise ValueError("d_true must be a square distance matrix.")
    if d_pred.shape[0] != d_pred.shape[1]:
        raise ValueError("d_pred must be a square distance matrix.")

    P = _distances_to_probabilities(d_true, sigma)
    Q = _distances_to_probabilities(d_pred, sigma)

    result: float = float(np.sum(P * np.log(P / Q)))
    return result