Skip to content

cca_zoo.nonparametric

Kernel- and graph-based nonparametric CCA methods.


KCCA

KCCA(
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object]
    | list[dict[str, object]]
    | None = None,
    eps: float = 0.001,
)

Bases: BaseModel

Kernel Canonical Correlation Analysis.

Extends MCCA to nonlinear relationships by mapping each view into a reproducing kernel Hilbert space via a kernel function \(k_i\). The dual variables (kernel coefficients) \(\boldsymbol{\alpha}_i\) are found by solving the kernelised generalised eigenvalue problem:

\[ A \boldsymbol{\alpha} = \lambda B \boldsymbol{\alpha} \]

where:

  • \(A\) is the between-kernel cross-covariance block matrix.
  • \(B = \mathrm{block\_diag}\bigl( c_i K_i + (1 - c_i) K_i^2 \bigr)\) is the regularised within-kernel matrix.
References

Hardoon, D. R., Szedmak, S., & Shawe-Taylor, J. (2004). Canonical correlation analysis: An overview with application to learning methods. Neural Computation, 16(12), 2639–2664.

Parameters:

Name Type Description Default
latent_dimensions int

Number of latent dimensions. Default is 1.

1
center bool

Whether to subtract column means before fitting. Default True.

True
c float | list[float]

Regularisation parameter(s) in [0, 1]. Default is 0.1.

0.1
kernel str | list[str]

Kernel name(s) or callable(s) passed to :func:sklearn.metrics.pairwise_kernels. Default is "linear".

'linear'
gamma float | list[float | None] | None

Gamma parameter(s) for the RBF/polynomial kernel.

None
degree float | list[float]

Degree parameter(s) for the polynomial kernel.

1.0
coef0 float | list[float]

coef0 parameter(s) for the polynomial/sigmoid kernel.

1.0
kernel_params dict[str, object] | list[dict[str, object]] | None

Extra per-view keyword arguments for the kernel.

None
eps float

Regularisation floor for the B matrix. Default is 1e-3.

0.001

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> X1 = rng.standard_normal((30, 5))
>>> X2 = rng.standard_normal((30, 5))
>>> model = KCCA(latent_dimensions=2, c=0.1).fit([X1, X2])
>>> scores = model.transform([X1, X2])
Source code in cca_zoo/nonparametric/_kcca.py
def __init__(
    self,
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object] | list[dict[str, object]] | None = None,
    eps: float = 1e-3,
) -> None:
    super().__init__(latent_dimensions=latent_dimensions, center=center)
    self.c = c
    self.kernel = kernel
    self.gamma = gamma
    self.degree = degree
    self.coef0 = coef0
    self.kernel_params = kernel_params
    self.eps = eps

fit

fit(views: list[ArrayLike], y: None = None) -> KCCA

Fit the KCCA model.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples, n_features_i).

required
y None

Ignored.

None

Returns:

Name Type Description
self KCCA

Fitted estimator.

Raises:

Type Description
ValueError

If fewer than 2 views are provided.

ValueError

If views have inconsistent numbers of samples.

Source code in cca_zoo/nonparametric/_kcca.py
def fit(self, views: list[ArrayLike], y: None = None) -> KCCA:
    """Fit the KCCA model.

    Args:
        views: List of arrays, each (n_samples, n_features_i).
        y: Ignored.

    Returns:
        self: Fitted estimator.

    Raises:
        ValueError: If fewer than 2 views are provided.
        ValueError: If views have inconsistent numbers of samples.
    """
    views_: list[np.ndarray] = self._setup_fit(views)
    c_ = perview_parameter("c", self.c, 0.1, self.n_views_)
    kernel_ = perview_parameter("kernel", self.kernel, "linear", self.n_views_)
    gamma_ = perview_parameter("gamma", self.gamma, None, self.n_views_)
    degree_ = perview_parameter("degree", self.degree, 1.0, self.n_views_)
    coef0_ = perview_parameter("coef0", self.coef0, 1.0, self.n_views_)
    kp_ = perview_parameter("kernel_params", self.kernel_params, {}, self.n_views_)

    self.train_views_: list[np.ndarray] = views_
    kernels = self._compute_kernels(views_, kernel_, gamma_, degree_, coef0_, kp_)
    A = self._build_A(kernels)
    B = self._build_B(kernels, c_)
    splits = np.cumsum([k.shape[1] for k in kernels])
    _, eigvecs = gevp(A, B, self.latent_dimensions)
    self.weights_: list[np.ndarray] = list(np.split(eigvecs, splits[:-1], axis=0))
    # Store kernel parameters for transform
    self._kernel: list[str] = kernel_
    self._gamma: list[float | None] = gamma_
    self._degree: list[float] = degree_
    self._coef0: list[float] = coef0_
    self._kp: list[dict[str, object]] = kp_
    return self

transform

transform(views: list[ArrayLike]) -> list[np.ndarray]

Transform new views using the fitted kernel dual variables.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples_test, n_features_i).

required

Returns:

Type Description
list[ndarray]

List of arrays, each (n_samples_test, latent_dimensions).

Raises:

Type Description
NotFittedError

If fit has not been called.

Source code in cca_zoo/nonparametric/_kcca.py
def transform(self, views: list[ArrayLike]) -> list[np.ndarray]:
    """Transform new views using the fitted kernel dual variables.

    Args:
        views: List of arrays, each (n_samples_test, n_features_i).

    Returns:
        List of arrays, each (n_samples_test, latent_dimensions).

    Raises:
        sklearn.exceptions.NotFittedError: If ``fit`` has not been called.
    """
    check_is_fitted(self)
    from cca_zoo._utils._validation import validate_views

    validated = validate_views(views)
    result = []
    for i, v in enumerate(validated):
        K_test = pairwise_kernels(
            self.train_views_[i],
            Y=v,
            metric=self._kernel[i],
            gamma=self._gamma[i],
            degree=self._degree[i],
            coef0=self._coef0[i],
            filter_params=True,
            **(self._kp[i] if self._kp[i] else {}),
        )
        result.append(K_test.T @ self.weights_[i])
    return result

KGCCA

KGCCA(
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object]
    | list[dict[str, object]]
    | None = None,
    view_weights: list[float] | None = None,
    eps: float = 1e-06,
)

Bases: BaseModel

Kernel Generalised Canonical Correlation Analysis.

Kernelised version of GCCA. The shared latent vector is found by solving the eigenvalue problem on the weighted sum of kernel projection matrices:

\[ Q = \sum_{i=1}^M \mu_i K_i \bigl(c_i K_i + (1 - c_i) K_i^2\bigr)^{-1} K_i \]

and the dual variables (kernel coefficients) are recovered as \(\boldsymbol{\alpha}_i = K_i^+ T\) where \(T\) is the matrix of top-k eigenvectors of \(Q\).

References

Tenenhaus, A., Philippe, C., & Frouin, V. (2015). Kernel generalized canonical correlation analysis. Computational Statistics & Data Analysis, 90, 114–131.

Parameters:

Name Type Description Default
latent_dimensions int

Number of latent dimensions. Default is 1.

1
center bool

Whether to subtract column means before fitting. Default True.

True
c float | list[float]

Regularisation parameter(s). Default is 0.1.

0.1
kernel str | list[str]

Kernel name(s). Default is "linear".

'linear'
gamma float | list[float | None] | None

Gamma for RBF/polynomial kernel.

None
degree float | list[float]

Degree for polynomial kernel.

1.0
coef0 float | list[float]

coef0 for polynomial/sigmoid kernel.

1.0
kernel_params dict[str, object] | list[dict[str, object]] | None

Extra per-view kernel keyword arguments.

None
view_weights list[float] | None

Per-view weights. Default is equal weights.

None
eps float

Regularisation floor. Default is 1e-6.

1e-06

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> X1 = rng.standard_normal((30, 5))
>>> X2 = rng.standard_normal((30, 5))
>>> X3 = rng.standard_normal((30, 5))
>>> model = KGCCA(latent_dimensions=2).fit([X1, X2, X3])
Source code in cca_zoo/nonparametric/_kgcca.py
def __init__(
    self,
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object] | list[dict[str, object]] | None = None,
    view_weights: list[float] | None = None,
    eps: float = 1e-6,
) -> None:
    super().__init__(latent_dimensions=latent_dimensions, center=center)
    self.c = c
    self.kernel = kernel
    self.gamma = gamma
    self.degree = degree
    self.coef0 = coef0
    self.kernel_params = kernel_params
    self.view_weights = view_weights
    self.eps = eps

fit

fit(views: list[ArrayLike], y: None = None) -> KGCCA

Fit the KGCCA model.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples, n_features_i).

required
y None

Ignored.

None

Returns:

Name Type Description
self KGCCA

Fitted estimator.

Raises:

Type Description
ValueError

If fewer than 2 views are provided.

ValueError

If views have inconsistent numbers of samples.

Source code in cca_zoo/nonparametric/_kgcca.py
def fit(self, views: list[ArrayLike], y: None = None) -> KGCCA:
    """Fit the KGCCA model.

    Args:
        views: List of arrays, each (n_samples, n_features_i).
        y: Ignored.

    Returns:
        self: Fitted estimator.

    Raises:
        ValueError: If fewer than 2 views are provided.
        ValueError: If views have inconsistent numbers of samples.
    """
    views_: list[np.ndarray] = self._setup_fit(views)
    c_ = perview_parameter("c", self.c, 0.1, self.n_views_)
    mu = perview_parameter("view_weights", self.view_weights, 1.0, self.n_views_)
    kernel_ = perview_parameter("kernel", self.kernel, "linear", self.n_views_)
    gamma_ = perview_parameter("gamma", self.gamma, None, self.n_views_)
    degree_ = perview_parameter("degree", self.degree, 1.0, self.n_views_)
    coef0_ = perview_parameter("coef0", self.coef0, 1.0, self.n_views_)
    kp_ = perview_parameter("kernel_params", self.kernel_params, {}, self.n_views_)

    self.train_views_: list[np.ndarray] = views_
    kernels = [
        pairwise_kernels(
            v,
            metric=kernel_[i],
            gamma=gamma_[i],
            degree=degree_[i],
            coef0=coef0_[i],
            filter_params=True,
            **(kp_[i] if kp_[i] else {}),
        )
        for i, v in enumerate(views_)
    ]
    # Build Q (n x n)
    Q = np.zeros((self.n_samples_, self.n_samples_))
    for i, (K, ci, mi) in enumerate(zip(kernels, c_, mu)):
        B_i = ci * K + (1.0 - ci) * K @ K
        min_eig = np.linalg.eigvalsh(B_i).min()
        if min_eig < self.eps:
            B_i += (self.eps - min_eig) * np.eye(B_i.shape[0])
        Q += mi * K @ np.linalg.inv(B_i) @ K

    _, eigvecs = gevp(Q, None, self.latent_dimensions)
    T = eigvecs[:, : self.latent_dimensions]
    self.weights_: list[np.ndarray] = [np.linalg.pinv(K) @ T for K in kernels]
    # Store kernel parameters for transform
    self._kernel = kernel_
    self._gamma = gamma_
    self._degree = degree_
    self._coef0 = coef0_
    self._kp = kp_
    return self

transform

transform(views: list[ArrayLike]) -> list[np.ndarray]

Transform new views using fitted kernel dual variables.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples_test, n_features_i).

required

Returns:

Type Description
list[ndarray]

List of arrays, each (n_samples_test, latent_dimensions).

Raises:

Type Description
NotFittedError

If fit has not been called.

Source code in cca_zoo/nonparametric/_kgcca.py
def transform(self, views: list[ArrayLike]) -> list[np.ndarray]:
    """Transform new views using fitted kernel dual variables.

    Args:
        views: List of arrays, each (n_samples_test, n_features_i).

    Returns:
        List of arrays, each (n_samples_test, latent_dimensions).

    Raises:
        sklearn.exceptions.NotFittedError: If ``fit`` has not been called.
    """
    check_is_fitted(self)
    from cca_zoo._utils._validation import validate_views

    validated = validate_views(views)
    result = []
    for i, v in enumerate(validated):
        K_test = pairwise_kernels(
            self.train_views_[i],
            Y=v,
            metric=self._kernel[i],
            gamma=self._gamma[i],
            degree=self._degree[i],
            coef0=self._coef0[i],
            filter_params=True,
            **(self._kp[i] if self._kp[i] else {}),
        )
        result.append(K_test.T @ self.weights_[i])
    return result

KTCCA

KTCCA(
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object]
    | list[dict[str, object]]
    | None = None,
    eps: float = 0.001,
    random_state: int | None = None,
)

Bases: BaseModel

Kernel Tensor Canonical Correlation Analysis.

Extends TCCA to nonlinear relationships by computing the cross-moment tensor from whitened kernel matrices rather than from the raw views. Each kernel matrix \(K_i\) is whitened using its regularised self-product, then PARAFAC is applied to the resulting cross-moment tensor.

References

Kim, T.-K., Wong, S.-F., & Cipolla, R. (2007). Tensor canonical correlation analysis for action classification. CVPR 2007. IEEE.

Parameters:

Name Type Description Default
latent_dimensions int

Number of latent dimensions. Default is 1.

1
center bool

Whether to subtract column means before fitting. Default True.

True
c float | list[float]

Regularisation parameter(s). Default is 0.1.

0.1
kernel str | list[str]

Kernel name(s). Default is "linear".

'linear'
gamma float | list[float | None] | None

Gamma for RBF/polynomial kernel.

None
degree float | list[float]

Degree for polynomial kernel.

1.0
coef0 float | list[float]

coef0 for polynomial/sigmoid kernel.

1.0
kernel_params dict[str, object] | list[dict[str, object]] | None

Extra per-view keyword arguments for the kernel.

None
eps float

Regularisation floor. Default is 1e-3.

0.001
random_state int | None

Seed for PARAFAC. Default is None.

None

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> X1 = rng.standard_normal((20, 5))
>>> X2 = rng.standard_normal((20, 5))
>>> X3 = rng.standard_normal((20, 5))
>>> model = KTCCA(latent_dimensions=1, random_state=0).fit([X1, X2, X3])
Source code in cca_zoo/nonparametric/_ktcca.py
def __init__(
    self,
    latent_dimensions: int = 1,
    center: bool = True,
    c: float | list[float] = 0.1,
    kernel: str | list[str] = "linear",
    gamma: float | list[float | None] | None = None,
    degree: float | list[float] = 1.0,
    coef0: float | list[float] = 1.0,
    kernel_params: dict[str, object] | list[dict[str, object]] | None = None,
    eps: float = 1e-3,
    random_state: int | None = None,
) -> None:
    super().__init__(latent_dimensions=latent_dimensions, center=center)
    self.c = c
    self.kernel = kernel
    self.gamma = gamma
    self.degree = degree
    self.coef0 = coef0
    self.kernel_params = kernel_params
    self.eps = eps
    self.random_state = random_state

fit

fit(views: list[ArrayLike], y: None = None) -> KTCCA

Fit the KTCCA model.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples, n_features_i).

required
y None

Ignored.

None

Returns:

Name Type Description
self KTCCA

Fitted estimator.

Raises:

Type Description
ValueError

If fewer than 2 views are provided.

ValueError

If views have inconsistent numbers of samples.

Source code in cca_zoo/nonparametric/_ktcca.py
def fit(self, views: list[ArrayLike], y: None = None) -> KTCCA:
    """Fit the KTCCA model.

    Args:
        views: List of arrays, each (n_samples, n_features_i).
        y: Ignored.

    Returns:
        self: Fitted estimator.

    Raises:
        ValueError: If fewer than 2 views are provided.
        ValueError: If views have inconsistent numbers of samples.
    """
    views_: list[np.ndarray] = self._setup_fit(views)
    c_ = perview_parameter("c", self.c, 0.1, self.n_views_)
    kernel_ = perview_parameter("kernel", self.kernel, "linear", self.n_views_)
    gamma_ = perview_parameter("gamma", self.gamma, None, self.n_views_)
    degree_ = perview_parameter("degree", self.degree, 1.0, self.n_views_)
    coef0_ = perview_parameter("coef0", self.coef0, 1.0, self.n_views_)
    kp_ = perview_parameter("kernel_params", self.kernel_params, {}, self.n_views_)

    self.train_views_: list[np.ndarray] = views_
    # Store parameters for transform
    self._kernel = kernel_
    self._gamma = gamma_
    self._degree = degree_
    self._coef0 = coef0_
    self._kp = kp_

    kernels = [
        pairwise_kernels(
            v,
            metric=kernel_[i],
            gamma=gamma_[i],
            degree=degree_[i],
            coef0=coef0_[i],
            filter_params=True,
            **(kp_[i] if kp_[i] else {}),
        )
        for i, v in enumerate(views_)
    ]
    whitened, self._cov_invsqrt = self._whiten_kernels(kernels, c_)

    # Build cross-moment tensor
    M: np.ndarray | None = None
    for i, wk in enumerate(whitened):
        if M is None:
            M = wk
        else:
            for _ in range(len(M.shape) - 1):
                wk = np.expand_dims(wk, 1)
            M = np.expand_dims(M, -1) @ wk
    assert M is not None
    M = np.mean(M, 0)

    tl.set_backend("numpy")
    parafac_result = parafac(
        M,
        self.latent_dimensions,
        verbose=False,
        random_state=self.random_state,
    )
    self.weights_: list[np.ndarray] = [
        self._cov_invsqrt[i] @ fac for i, fac in enumerate(parafac_result.factors)
    ]
    return self

transform

transform(views: list[ArrayLike]) -> list[np.ndarray]

Transform new views using fitted kernel dual variables.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each (n_samples_test, n_features_i).

required

Returns:

Type Description
list[ndarray]

List of arrays, each (n_samples_test, latent_dimensions).

Raises:

Type Description
NotFittedError

If fit has not been called.

Source code in cca_zoo/nonparametric/_ktcca.py
def transform(self, views: list[ArrayLike]) -> list[np.ndarray]:
    """Transform new views using fitted kernel dual variables.

    Args:
        views: List of arrays, each (n_samples_test, n_features_i).

    Returns:
        List of arrays, each (n_samples_test, latent_dimensions).

    Raises:
        sklearn.exceptions.NotFittedError: If ``fit`` has not been called.
    """
    check_is_fitted(self)
    from cca_zoo._utils._validation import validate_views

    validated = validate_views(views)
    result = []
    for i, v in enumerate(validated):
        K_test = pairwise_kernels(
            self.train_views_[i],
            Y=v,
            metric=self._kernel[i],
            gamma=self._gamma[i],
            degree=self._degree[i],
            coef0=self._coef0[i],
            filter_params=True,
            **(self._kp[i] if self._kp[i] else {}),
        )
        result.append(K_test.T @ self.weights_[i])
    return result

ManifoldCCA

ManifoldCCA(
    latent_dimensions: int = 1,
    center: bool = True,
    method: str = "laplacian",
    n_neighbors: int | list[int] = 10,
    affinity: str | list[str] = "nearest_neighbors",
    gamma: float | list[float | None] | None = None,
    lle_reg: float | list[float] = 0.001,
    n_operator_components: int
    | list[int | None]
    | None = None,
    eps: float = 1e-06,
)

Bases: BaseModel

ManifoldCCA -- transductive multiview CCA over a shared manifold operator.

Every other multiview method in this library maximises cross-view covariance subject to a within-view covariance constraint (:class:~cca_zoo.linear.MCCA's ridge-blended sample covariance, :class:~cca_zoo.linear.GraphicalLassoCCA's sparse-precision covariance, :class:~cca_zoo.nonparametric.KCCA's regularised kernel Gram matrix). Spectral manifold-learning methods (:class:sklearn.manifold.SpectralEmbedding, :class:sklearn.manifold.LocallyLinearEmbedding) instead constrain a single view's embedding against a graph operator \(M\) built from that view's own local neighbourhood structure -- the graph Laplacian (\(M = D - W\), small \(\operatorname{tr}(Y^\top M Y)\) means neighbouring points map to nearby embeddings) or the LLE reconstruction operator (\(M = (I-W)^\top(I-W)\), small \(\operatorname{tr}(Y^\top M Y)\) means each point's embedding is well reconstructed from its neighbours'). Both are themselves generalised eigenproblems of exactly the same "maximise-subject-to-a-quadratic-constraint" shape as :class:~cca_zoo.linear.MCCA -- just with \(M\) replacing a covariance matrix, and with no covariance available at all in the usual sense, since there's no feature map: the "weight" is the per-training-point embedding.

ManifoldCCA solves the resulting joint, multiview version:

\[ \max_{Z_1, \dots, Z_M} \sum_{i \neq j} \operatorname{Cov}(Z_i, Z_j) \quad \text{subject to} \quad Z_i^\top M_i Z_i = I \; \forall i \]

where \(Z_i \in \mathbb{R}^{n \times k}\) is view \(i\)'s embedding of the training points and \(M_i\) is that view's own graph operator -- the same joint-eigenproblem construction :class:~cca_zoo.linear.MCCA and :class:~cca_zoo.nonparametric.KCCA use, but with each view's "feature map" being the identity (so the projection weight found by the solver is \(Z_i\) directly -- see :func:_centering_matrix) and its within-view block \(M_i\) instead of a covariance.

Solved after projecting both sides onto \(\mathbf{1}^\perp\) (see :func:_orthonormal_complement_of_ones): every \(M_i\) has the constant vector in its (near-)null space, as does the reward, so leaving it in would put a spurious, numerically unstable 0/0-type direction at the top of the spectrum -- the same reason :class:~sklearn.manifold.SpectralEmbedding always discards its own trivial constant solution. One consequence worth checking directly: if two views are given identical data, \(A\) (via the shared reward) restricted to \(\mathbf{1}^\perp\) is proportional to the identity, so the joint problem's solution for that view is exactly the ordinary Rayleigh-quotient minimiser of \(M_i\) alone -- i.e. it reduces exactly to plain single-view spectral embedding of that view, which is the right sanity check for "is this actually the natural multiview generalisation" (verified directly against :class:~sklearn.manifold.SpectralEmbedding's own affinity construction in the tests).

Since \(Z_i\) is only ever defined at the training points (there is no feature map to apply to new data), out-of-sample projection reuses each method's own established extension rather than a generic auxiliary model bolted on afterward:

  • method="lle": a new point's barycentric reconstruction weights (see :func:_barycenter_weights) against its n_neighbors nearest training points, applied directly to those training points' rows of the fitted \(Z_i\) -- exactly :meth:~sklearn.manifold.LocallyLinearEmbedding.transform's own mechanism, reusing the identical weight computation training used. Valid because that formula is linear in the training embedding: with \(Z_i = Y_i B_i\) (\(Y_i\) the per-view smooth basis, \(B_i\) the joint solve's combination -- see :meth:fit), applying new-point weights to \(Y_i\) first and then \(B_i\), or to \(Z_i = Y_i B_i\) directly, are the same computation.
  • method="laplacian": the classical Nystrom extension (Bengio et al. 2003) of each kept eigenvector individually -- \(y_k(x) = \tfrac{1}{\mu_k}\sum_j \tilde{W}(x, x_j)\, y_k(x_j)\), \(\tilde W\) the degree-normalised new-to-training affinity (built by the same rule as the training graph, see :func:_laplacian_new_point_affinity) and \(\mu_k = 1 - \lambda_k\) -- then the same \(B_i\) combination used at training time. Unlike the LLE case, this can't collapse to a single "apply weights to \(Z_i\)" step, since each eigenvector has its own \(\mu_k\) rescaling before \(B_i\) mixes them.

Both are exact consequences of what :meth:fit actually solved, not a separately-fit approximation of it.

Note

Unlike :class:~sklearn.manifold.LocallyLinearEmbedding, Hessian-LLE (method="hessian") and LTSA (method="ltsa") are not implemented here -- both need a local Hessian/tangent-space estimate per point (local PCA plus a polynomial basis, then a null-space projection) that's substantially more involved to get right than the graph Laplacian or LLE's own reconstruction weights, and are left for a future extension rather than shipped undertested. Isomap-flavoured (geodesic-distance) regularisation is achievable today via :class:~cca_zoo.nonparametric.KCCA with a precomputed geodesic Gram matrix in place of a standard kernel, so isn't duplicated here.

inverse_transform/predict are not supported (same as :class:~cca_zoo.nonparametric.KCCA): both rely on BaseModel's default view-loading fit, which assumes weights_[i] has shape (n_features_i, k); here it is (n_train_samples, k) (the training embedding itself), since there is no feature-space weight vector to speak of.

method itself is not a per-view parameter, unlike every other constructor argument: mixing "laplacian" and "lle" across views is not mathematically ruled out (the joint eigenproblem in :meth:fit only ever consumes each view's own basis/eigenvalues, regardless of which operator produced them), but transform's out-of-sample extension dispatches on method once for every view at once, and would need its own per-view branch and per-view-typed fitted state to support a genuine mix -- a larger, separate change from exposing this class's already per-view-independent operator hyperparameters.

Solves an \((nM) \times (nM)\) dense generalised eigenproblem (\(n\) = training samples, \(M\) = number of views), the same cost profile as :class:~cca_zoo.nonparametric.KCCA -- intended for moderate training-set sizes, not the very large-\(n\) regime.

References

Roweis, S. T., & Saul, L. K. (2000). Nonlinear dimensionality reduction by locally linear embedding. Science, 290(5500), 2323-2326.

Belkin, M., & Niyogi, P. (2003). Laplacian eigenmaps for dimensionality reduction and data representation. Neural Computation, 15(6), 1373-1396.

Parameters:

Name Type Description Default
latent_dimensions int

Number of latent dimensions. Default is 1.

1
center bool

Whether to subtract column means before fitting. Default True.

True
method str

"laplacian" (graph Laplacian, matching :class:~sklearn.manifold.SpectralEmbedding) or "lle" (locally linear embedding operator), the same for every view. Default "laplacian".

'laplacian'
n_neighbors int | list[int]

Number of neighbours used to build each view's graph. Either a single value applied to every view or a list of per-view values. Default 10.

10
affinity str | list[str]

"nearest_neighbors" or "rbf", passed to :class:~sklearn.manifold.SpectralEmbedding when method="laplacian". Ignored for method="lle". Either a single value applied to every view or a list of per-view values. Default "nearest_neighbors".

'nearest_neighbors'
gamma float | list[float | None] | None

RBF kernel coefficient(s), used only when method="laplacian" and affinity="rbf". Either a single float (or None) applied to every view or a list of per-view values. Default None (sklearn's own 1 / n_features default).

None
lle_reg float | list[float]

Regularisation added to each point's local reconstruction Gram matrix, used only when method="lle". Either a single float applied to every view or a list of per-view floats. Default 1e-3.

0.001
n_operator_components int | list[int | None] | None

Number of each view's own smallest-eigenvalue operator components kept before the joint eigenproblem is solved (see :func:_smooth_basis) -- effectively this class's regularisation strength, the same role c plays elsewhere in this library, just the opposite direction: smaller keeps fewer, "smoother" candidate directions per view (more regularised, more denoised, but liable to discard real structure if set too small); the untruncated limit (n_operator_components = n_samples - 1) hands each view as many free directions as there are training points, which -- like any unregularised multivariate CCA at that dimensionality-to-sample-size ratio -- fabricates spurious cross-view correlation out of pure noise (see tests/nonparametric/test_manifold_cca.py's comparison against plain unregularised MCCA at the same nominal dimensionality). Either a single value (or None) applied to every view or a list of per-view values. Default None: max(4 * latent_dimensions, 10), clipped to n_samples - 1, independently per view.

None
eps float

Floor applied to each kept operator eigenvalue (see :func:_smooth_basis) to ensure positive definiteness. Default 1e-6.

1e-06

Examples:

>>> import numpy as np
>>> rng = np.random.default_rng(0)
>>> X1 = rng.standard_normal((60, 8))
>>> X2 = rng.standard_normal((60, 6))
>>> model = ManifoldCCA(method="laplacian", n_neighbors=8).fit([X1, X2])
>>> scores = model.transform([X1, X2])

A different neighbourhood size and regularisation strength per view:

>>> model = ManifoldCCA(
...     method="laplacian", n_neighbors=[8, 12], n_operator_components=[10, 15]
... ).fit([X1, X2])
Source code in cca_zoo/nonparametric/_manifold_cca.py
def __init__(
    self,
    latent_dimensions: int = 1,
    center: bool = True,
    method: str = "laplacian",
    n_neighbors: int | list[int] = 10,
    affinity: str | list[str] = "nearest_neighbors",
    gamma: float | list[float | None] | None = None,
    lle_reg: float | list[float] = 1e-3,
    n_operator_components: int | list[int | None] | None = None,
    eps: float = 1e-6,
) -> None:
    super().__init__(latent_dimensions=latent_dimensions, center=center)
    self.method = method
    self.n_neighbors = n_neighbors
    self.affinity = affinity
    self.gamma = gamma
    self.lle_reg = lle_reg
    self.n_operator_components = n_operator_components
    self.eps = eps

fit

fit(views: list[ArrayLike], y: None = None) -> ManifoldCCA

Fit ManifoldCCA by a joint generalised eigenproblem over per-view graphs.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of 2 or more arrays, each (n_samples, n_features_i).

required
y None

Ignored.

None

Returns:

Name Type Description
self ManifoldCCA

Fitted estimator.

Raises:

Type Description
ValueError

If fewer than 2 views are provided.

ValueError

If views have inconsistent numbers of samples.

Source code in cca_zoo/nonparametric/_manifold_cca.py
def fit(self, views: list[ArrayLike], y: None = None) -> ManifoldCCA:
    """Fit ManifoldCCA by a joint generalised eigenproblem over per-view graphs.

    Args:
        views: List of 2 or more arrays, each (n_samples, n_features_i).
        y: Ignored.

    Returns:
        self: Fitted estimator.

    Raises:
        ValueError: If fewer than 2 views are provided.
        ValueError: If views have inconsistent numbers of samples.
    """
    views_ = self._setup_fit(views)
    n = self.n_samples_
    m = self.n_views_

    n_neighbors_ = perview_parameter("n_neighbors", self.n_neighbors, 10, m)
    affinity_ = perview_parameter("affinity", self.affinity, "nearest_neighbors", m)
    gamma_ = perview_parameter("gamma", self.gamma, None, m)
    lle_reg_ = perview_parameter("lle_reg", self.lle_reg, 1e-3, m)
    n_operator_components_ = perview_parameter(
        "n_operator_components", self.n_operator_components, None, m
    )

    # Project onto the constant vector's orthogonal complement first --
    # see _orthonormal_complement_of_ones -- so the shared (near-)null
    # direction every operator and the reward both have never enters
    # the solve at all, rather than being merely floored.
    P = _orthonormal_complement_of_ones(n)
    C_reduced = P.T @ _centering_matrix(n) @ P

    k_ops = [
        self._resolve_n_operator_components(c, n) for c in n_operator_components_
    ]
    bases = []
    full_bases = []
    eigenvalue_blocks = []
    laplacian_degrees: list[np.ndarray] = []
    laplacian_gamma: list[float | None] = []
    laplacian_nn: list[NearestNeighbors | None] = []
    lle_nn: list[NearestNeighbors] = []
    for v, nn_i, aff_i, gamma_i, lle_reg_i, k_op in zip(
        views_, n_neighbors_, affinity_, gamma_, lle_reg_, k_ops
    ):
        if self.method == "laplacian":
            W, resolved_gamma = _laplacian_affinity(v, nn_i, aff_i, gamma_i)
            operator = _normalised_laplacian(W)
            laplacian_degrees.append(W.sum(axis=1))
            laplacian_gamma.append(resolved_gamma)
            laplacian_nn.append(
                NearestNeighbors(n_neighbors=nn_i).fit(v)
                if aff_i == "nearest_neighbors"
                else None
            )
        else:
            operator = _lle_operator(v, nn_i, lle_reg_i)
            lle_nn.append(NearestNeighbors(n_neighbors=nn_i + 1).fit(v))

        reduced_operator = P.T @ operator @ P
        basis, eigenvalues = _smooth_basis(reduced_operator, k_op, self.eps)
        bases.append(basis)
        full_bases.append(P @ basis)
        eigenvalue_blocks.append(eigenvalues)

    offsets = np.concatenate([[0], np.cumsum(k_ops)])
    B = np.asarray(block_diag(*[np.diag(ev) for ev in eigenvalue_blocks])) / m
    A = np.zeros((offsets[-1], offsets[-1]))
    for i in range(m):
        for j in range(m):
            if i != j:
                block = bases[i].T @ C_reduced @ bases[j]
                A[offsets[i] : offsets[i + 1], offsets[j] : offsets[j + 1]] = block
    A /= m

    _, eigvecs = gevp(A, B, self.latent_dimensions)
    blocks = list(np.split(eigvecs, offsets[1:-1], axis=0))
    embedding = [fb @ blk for fb, blk in zip(full_bases, blocks)]
    self.weights_: list[np.ndarray] = embedding
    self._n_neighbors_: list[int] = n_neighbors_
    self._affinity_: list[str] = affinity_
    self._lle_reg_: list[float] = lle_reg_

    if self.method == "laplacian":
        self._laplacian_state_: list[_LaplacianViewState] | None = [
            _LaplacianViewState(
                full_basis=full_bases[i],
                mu=np.maximum(1.0 - eigenvalue_blocks[i], _NYSTROM_MU_FLOOR),
                block=blocks[i],
                degrees=laplacian_degrees[i],
                gamma=laplacian_gamma[i],
                nn=laplacian_nn[i],
            )
            for i in range(m)
        ]
        self._lle_state_: list[NearestNeighbors] | None = None
    else:
        self._laplacian_state_ = None
        self._lle_state_ = lle_nn
    return self

transform

transform(views: list[ArrayLike]) -> list[np.ndarray]

Project new views via each method's own out-of-sample extension.

method="lle" reuses :func:_barycenter_weights against each new point's nearest training points (exactly :meth:~sklearn.manifold.LocallyLinearEmbedding.transform's own mechanism); method="laplacian" uses the classical Nystrom extension of each kept eigenvector (Bengio et al. 2003) followed by the same combination :meth:fit used. See the class docstring.

Parameters:

Name Type Description Default
views list[ArrayLike]

List of arrays, each of shape (n_samples, n_features_i).

required

Returns:

Type Description
list[ndarray]

List of arrays, each of shape (n_samples, latent_dimensions).

Raises:

Type Description
NotFittedError

If fit has not been called.

Source code in cca_zoo/nonparametric/_manifold_cca.py
def transform(self, views: list[ArrayLike]) -> list[np.ndarray]:
    """Project new views via each method's own out-of-sample extension.

    ``method="lle"`` reuses :func:`_barycenter_weights` against each new
    point's nearest training points (exactly
    :meth:`~sklearn.manifold.LocallyLinearEmbedding.transform`'s own
    mechanism); ``method="laplacian"`` uses the classical Nystrom
    extension of each kept eigenvector (Bengio et al. 2003) followed by
    the same combination :meth:`fit` used. See the class docstring.

    Args:
        views: List of arrays, each of shape (n_samples, n_features_i).

    Returns:
        List of arrays, each of shape (n_samples, latent_dimensions).

    Raises:
        sklearn.exceptions.NotFittedError: If ``fit`` has not been called.
    """
    check_is_fitted(self)
    validated = validate_views(views)
    centred = [v - m for v, m in zip(validated, self.means_)]

    if self.method == "lle":
        assert self._lle_state_ is not None
        result = []
        for v_new, v_train, nn, z, nn_i, lle_reg_i in zip(
            centred,
            self._views_fit_,
            self._lle_state_,
            self.weights_,
            self._n_neighbors_,
            self._lle_reg_,
        ):
            indices = nn.kneighbors(v_new, n_neighbors=nn_i, return_distance=False)
            weights = _barycenter_weights(v_new, v_train, indices, lle_reg_i)
            result.append(np.einsum("qn,qnk->qk", weights, z[indices]))
        return result

    assert self._laplacian_state_ is not None
    result = []
    for v_new, v_train, state, aff_i, nn_i in zip(
        centred,
        self._views_fit_,
        self._laplacian_state_,
        self._affinity_,
        self._n_neighbors_,
    ):
        W_new = _laplacian_new_point_affinity(
            v_new,
            v_train,
            aff_i,
            state.gamma,
            nn_i,
            state.nn,
        )
        degrees_new = np.maximum(W_new.sum(axis=1), 1e-12)
        K_tilde = W_new / np.sqrt(np.outer(degrees_new, state.degrees))
        extended = (K_tilde @ state.full_basis) / state.mu
        result.append(extended @ state.block)
    return result