Skip to content

cca_zoo.metrics

Metrics for evaluating fitted multiview CCA models. Functions take already-computed arrays (latent scores, loadings, a correlation matrix) rather than a fitted model, the same convention sklearn.metrics uses.


pairwise_correlations

pairwise_correlations(
    transformed: Sequence[ArrayLike],
) -> np.ndarray

Full pairwise Pearson correlation matrix between views' latent scores.

Parameters:

Name Type Description Default
transformed Sequence[ArrayLike]

List of arrays, each of shape (n_samples, latent_dimensions) -- one per view's own canonical variate (e.g. the output of a fitted model's transform).

required

Returns:

Type Description
ndarray

Array of shape (n_views, n_views, latent_dimensions) where entry

ndarray

[i, j, d] is the Pearson correlation between view i's and view

ndarray

j's d-th canonical variate.

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
>>> corrs = pairwise_correlations([t1, t2])
>>> corrs.shape
(2, 2, 1)
>>> round(float(corrs[0, 1, 0]), 2)
0.98
Source code in cca_zoo/metrics/_correlation.py
def pairwise_correlations(transformed: Sequence[ArrayLike]) -> np.ndarray:
    """Full pairwise Pearson correlation matrix between views' latent scores.

    Args:
        transformed: List of arrays, each of shape (n_samples,
            latent_dimensions) -- one per view's own canonical variate
            (e.g. the output of a fitted model's ``transform``).

    Returns:
        Array of shape (n_views, n_views, latent_dimensions) where entry
        ``[i, j, d]`` is the Pearson correlation between view i's and view
        j's d-th canonical variate.

    Examples:
        >>> import numpy as np
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
        >>> corrs = pairwise_correlations([t1, t2])
        >>> corrs.shape
        (2, 2, 1)
        >>> round(float(corrs[0, 1, 0]), 2)
        0.98
    """
    T = np.stack([np.asarray(t) for t in transformed], axis=0)
    T = T - T.mean(axis=1, keepdims=True)
    norms = np.sqrt((T**2).sum(axis=1, keepdims=True))
    T_norm = T / np.where(norms > 1e-12, norms, 1.0)
    corrs: np.ndarray = np.einsum("isd,jsd->ijd", T_norm, T_norm)
    return corrs

average_pairwise_correlations

average_pairwise_correlations(
    correlations: ArrayLike,
) -> np.ndarray

Mean off-diagonal pairwise correlation per canonical dimension.

Parameters:

Name Type Description Default
correlations ArrayLike

Array of shape (n_views, n_views, latent_dimensions) -- typically the output of :func:pairwise_correlations.

required

Returns:

Type Description
ndarray

Array of shape (latent_dimensions,) with the average off-diagonal

ndarray

pairwise correlation for each canonical dimension.

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
>>> corrs = pairwise_correlations([t1, t2])
>>> avg = average_pairwise_correlations(corrs)
>>> avg.shape
(1,)
>>> round(float(avg[0]), 2)
0.98
Source code in cca_zoo/metrics/_correlation.py
def average_pairwise_correlations(correlations: ArrayLike) -> np.ndarray:
    """Mean off-diagonal pairwise correlation per canonical dimension.

    Args:
        correlations: Array of shape (n_views, n_views, latent_dimensions)
            -- typically the output of :func:`pairwise_correlations`.

    Returns:
        Array of shape (latent_dimensions,) with the average off-diagonal
        pairwise correlation for each canonical dimension.

    Examples:
        >>> import numpy as np
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
        >>> corrs = pairwise_correlations([t1, t2])
        >>> avg = average_pairwise_correlations(corrs)
        >>> avg.shape
        (1,)
        >>> round(float(avg[0]), 2)
        0.98
    """
    corrs = np.asarray(correlations)
    n_views = corrs.shape[0]
    off_diag_sum: np.ndarray = corrs.sum(axis=(0, 1)) - sum(
        corrs[i, i, :] for i in range(n_views)
    )
    n_pairs = n_views * (n_views - 1)
    result: np.ndarray = off_diag_sum / n_pairs
    return result

factor_loadings

factor_loadings(
    views: Sequence[ArrayLike],
    transformed: Sequence[ArrayLike],
) -> list[np.ndarray]

Pearson correlation between each raw feature and its view's own variate.

Parameters:

Name Type Description Default
views Sequence[ArrayLike]

List of arrays, each of shape (n_samples, n_features_i).

required
transformed Sequence[ArrayLike]

List of arrays, each of shape (n_samples, latent_dimensions), aligned with views -- view i's own canonical variate.

required

Returns:

Type Description
list[ndarray]

List of arrays, each of shape (n_features_i, latent_dimensions),

list[ndarray]

where entry [j, d] is the correlation between feature j of

list[ndarray]

view i and the d-th canonical variate of view i.

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> view1 = np.column_stack(
...     [t1[:, 0] + 0.3 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> loadings = factor_loadings([view1], [t1])
>>> loadings[0].shape
(2, 1)
>>> round(float(loadings[0][0, 0]), 2)
0.97
Source code in cca_zoo/metrics/_correlation.py
def factor_loadings(
    views: Sequence[ArrayLike], transformed: Sequence[ArrayLike]
) -> list[np.ndarray]:
    """Pearson correlation between each raw feature and its view's own variate.

    Args:
        views: List of arrays, each of shape (n_samples, n_features_i).
        transformed: List of arrays, each of shape (n_samples,
            latent_dimensions), aligned with ``views`` -- view i's own
            canonical variate.

    Returns:
        List of arrays, each of shape (n_features_i, latent_dimensions),
        where entry ``[j, d]`` is the correlation between feature j of
        view i and the d-th canonical variate of view i.

    Examples:
        >>> import numpy as np
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> view1 = np.column_stack(
        ...     [t1[:, 0] + 0.3 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> loadings = factor_loadings([view1], [t1])
        >>> loadings[0].shape
        (2, 1)
        >>> round(float(loadings[0][0, 0]), 2)
        0.97
    """
    loadings = []
    for view, variate in zip(views, transformed):
        v = np.asarray(view)
        t = np.asarray(variate)
        v_c = v - v.mean(axis=0)
        t_c = t - t.mean(axis=0)
        cov = v_c.T @ t_c / (v.shape[0] - 1)
        std_v = np.maximum(v_c.std(axis=0, ddof=1), 1e-12)
        std_t = np.maximum(t_c.std(axis=0, ddof=1), 1e-12)
        loadings.append(cov / np.outer(std_v, std_t))
    return loadings

adequacy_coefficient

adequacy_coefficient(
    loadings: Sequence[ArrayLike],
) -> list[np.ndarray]

Own-set variance each view's canonical variates extract from that view.

Also known as the adequacy coefficient (Cramer & Nicewander, 1979) or per-dimension communality: the mean squared factor loading, i.e. the average proportion of a view's own feature variance captured by each of its canonical variates.

Parameters:

Name Type Description Default
loadings Sequence[ArrayLike]

List of arrays, each of shape (n_features_i, latent_dimensions) -- typically the output of :func:~cca_zoo.metrics.factor_loadings.

required

Returns:

Type Description
list[ndarray]

List of arrays, each of shape (latent_dimensions,): view i's own

list[ndarray]

variance extracted by each of its canonical dimensions.

Examples:

>>> import numpy as np
>>> from cca_zoo.metrics import factor_loadings
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> view1 = np.column_stack(
...     [t1[:, 0] + 0.3 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> loadings = factor_loadings([view1], [t1])
>>> adequacy = adequacy_coefficient(loadings)
>>> adequacy[0].shape
(1,)
>>> round(float(adequacy[0][0]), 2)
0.48
Source code in cca_zoo/metrics/_redundancy.py
def adequacy_coefficient(loadings: Sequence[ArrayLike]) -> list[np.ndarray]:
    """Own-set variance each view's canonical variates extract from that view.

    Also known as the adequacy coefficient (Cramer & Nicewander, 1979) or
    per-dimension communality: the mean squared factor loading, i.e. the
    average proportion of a view's own feature variance captured by each
    of its canonical variates.

    Args:
        loadings: List of arrays, each of shape (n_features_i,
            latent_dimensions) -- typically the output of
            :func:`~cca_zoo.metrics.factor_loadings`.

    Returns:
        List of arrays, each of shape (latent_dimensions,): view i's own
        variance extracted by each of its canonical dimensions.

    Examples:
        >>> import numpy as np
        >>> from cca_zoo.metrics import factor_loadings
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> view1 = np.column_stack(
        ...     [t1[:, 0] + 0.3 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> loadings = factor_loadings([view1], [t1])
        >>> adequacy = adequacy_coefficient(loadings)
        >>> adequacy[0].shape
        (1,)
        >>> round(float(adequacy[0][0]), 2)
        0.48
    """
    return [np.mean(np.asarray(loading) ** 2, axis=0) for loading in loadings]

redundancy_index

redundancy_index(
    loadings: Sequence[ArrayLike], correlations: ArrayLike
) -> np.ndarray

Stewart & Love (1968) redundancy: variance in view i explained via view j.

For each ordered pair of views and canonical dimension, the proportion of view i's own variance that is both captured by its d-th canonical variate (:func:adequacy_coefficient) and shared with view j (the squared canonical correlation between their d-th variates). Unlike a canonical correlation, redundancy is asymmetric: view i's redundancy given view j need not equal view j's redundancy given view i, since each view's own adequacy can differ.

Parameters:

Name Type Description Default
loadings Sequence[ArrayLike]

List of arrays, each of shape (n_features_i, latent_dimensions) -- typically the output of :func:~cca_zoo.metrics.factor_loadings.

required
correlations ArrayLike

Array of shape (n_views, n_views, latent_dimensions) -- typically the output of :func:~cca_zoo.metrics.pairwise_correlations.

required

Returns:

Type Description
ndarray

Array of shape (n_views, n_views, latent_dimensions), where entry

ndarray

[i, j, d] is the redundancy of view i given view j's d-th

ndarray

canonical variate. The diagonal [i, i, d] equals view i's own

ndarray

adequacy coefficient, since a view's correlation with itself is 1.

Examples:

>>> import numpy as np
>>> from cca_zoo.metrics import factor_loadings, pairwise_correlations
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
>>> view1 = np.column_stack(
...     [t1[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> view2 = np.column_stack(
...     [t2[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> loadings = factor_loadings([view1, view2], [t1, t2])
>>> corrs = pairwise_correlations([t1, t2])
>>> redundancy = redundancy_index(loadings, corrs)
>>> redundancy.shape
(2, 2, 1)
>>> round(float(redundancy[0, 1, 0]), 2)
0.52
Source code in cca_zoo/metrics/_redundancy.py
def redundancy_index(
    loadings: Sequence[ArrayLike], correlations: ArrayLike
) -> np.ndarray:
    """Stewart & Love (1968) redundancy: variance in view i explained via view j.

    For each ordered pair of views and canonical dimension, the proportion
    of view i's own variance that is both captured by its d-th canonical
    variate (:func:`adequacy_coefficient`) *and* shared with view j (the
    squared canonical correlation between their d-th variates). Unlike a
    canonical correlation, redundancy is asymmetric: view i's redundancy
    given view j need not equal view j's redundancy given view i, since
    each view's own adequacy can differ.

    Args:
        loadings: List of arrays, each of shape (n_features_i,
            latent_dimensions) -- typically the output of
            :func:`~cca_zoo.metrics.factor_loadings`.
        correlations: Array of shape (n_views, n_views, latent_dimensions)
            -- typically the output of
            :func:`~cca_zoo.metrics.pairwise_correlations`.

    Returns:
        Array of shape (n_views, n_views, latent_dimensions), where entry
        ``[i, j, d]`` is the redundancy of view i given view j's d-th
        canonical variate. The diagonal ``[i, i, d]`` equals view i's own
        adequacy coefficient, since a view's correlation with itself is 1.

    Examples:
        >>> import numpy as np
        >>> from cca_zoo.metrics import factor_loadings, pairwise_correlations
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
        >>> view1 = np.column_stack(
        ...     [t1[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> view2 = np.column_stack(
        ...     [t2[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> loadings = factor_loadings([view1, view2], [t1, t2])
        >>> corrs = pairwise_correlations([t1, t2])
        >>> redundancy = redundancy_index(loadings, corrs)
        >>> redundancy.shape
        (2, 2, 1)
        >>> round(float(redundancy[0, 1, 0]), 2)
        0.52
    """
    adequacy = np.stack(adequacy_coefficient(loadings), axis=0)  # (n_views, k)
    corrs = np.asarray(correlations)
    result: np.ndarray = adequacy[:, np.newaxis, :] * corrs**2
    return result

total_redundancy

total_redundancy(redundancy: ArrayLike) -> np.ndarray

Cumulative redundancy across every retained canonical dimension.

Parameters:

Name Type Description Default
redundancy ArrayLike

Array of shape (n_views, n_views, latent_dimensions) -- typically the output of :func:redundancy_index.

required

Returns:

Type Description
ndarray

Array of shape (n_views, n_views): entry [i, j] is the total

ndarray

proportion of view i's variance explained by view j's canonical

ndarray

variates, summed over every retained dimension.

Examples:

>>> import numpy as np
>>> from cca_zoo.metrics import factor_loadings, pairwise_correlations
>>> rng = np.random.default_rng(0)
>>> t1 = rng.standard_normal((20, 1))
>>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
>>> view1 = np.column_stack(
...     [t1[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> view2 = np.column_stack(
...     [t2[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
... )
>>> loadings = factor_loadings([view1, view2], [t1, t2])
>>> corrs = pairwise_correlations([t1, t2])
>>> redundancy = redundancy_index(loadings, corrs)
>>> total = total_redundancy(redundancy)
>>> total.shape
(2, 2)
>>> round(float(total[0, 1]), 2)
0.52
Source code in cca_zoo/metrics/_redundancy.py
def total_redundancy(redundancy: ArrayLike) -> np.ndarray:
    """Cumulative redundancy across every retained canonical dimension.

    Args:
        redundancy: Array of shape (n_views, n_views, latent_dimensions) --
            typically the output of :func:`redundancy_index`.

    Returns:
        Array of shape (n_views, n_views): entry ``[i, j]`` is the total
        proportion of view i's variance explained by view j's canonical
        variates, summed over every retained dimension.

    Examples:
        >>> import numpy as np
        >>> from cca_zoo.metrics import factor_loadings, pairwise_correlations
        >>> rng = np.random.default_rng(0)
        >>> t1 = rng.standard_normal((20, 1))
        >>> t2 = 0.8 * t1 + 0.2 * rng.standard_normal((20, 1))
        >>> view1 = np.column_stack(
        ...     [t1[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> view2 = np.column_stack(
        ...     [t2[:, 0] + 0.1 * rng.standard_normal(20), rng.standard_normal(20)]
        ... )
        >>> loadings = factor_loadings([view1, view2], [t1, t2])
        >>> corrs = pairwise_correlations([t1, t2])
        >>> redundancy = redundancy_index(loadings, corrs)
        >>> total = total_redundancy(redundancy)
        >>> total.shape
        (2, 2)
        >>> round(float(total[0, 1]), 2)
        0.52
    """
    result: np.ndarray = np.asarray(redundancy).sum(axis=-1)
    return result