# Paper-scoped implementation; see SOURCE_PROVENANCE.json. from __future__ import annotations from typing import Any import numpy as np from numpy.typing import NDArray def global_classes(y: Any, n_classes: int | None=None) -> NDArray[np.int64]: yy = np.asarray(y, dtype=np.int64) if yy.size == 0: raise ValueError('Cannot infer classes from an empty target array') if np.min(yy) < 0: raise ValueError('Class labels must be non-negative integer IDs') k = int(n_classes) if n_classes is not None else int(np.max(yy)) + 1 if k <= int(np.max(yy)): raise ValueError(f'n_classes={k} is incompatible with maximum label {int(np.max(yy))}') return np.arange(k, dtype=np.int64) def estimator_classes(estimator: Any) -> NDArray[np.int64]: classes = getattr(estimator, 'classes_', None) if classes is None and hasattr(estimator, 'named_steps'): final = list(estimator.named_steps.values())[-1] classes = getattr(final, 'classes_', None) if classes is None: raise AttributeError('Estimator does not expose classes_; probability columns cannot be aligned safely') return np.asarray(classes, dtype=np.int64) def decision_to_proba(decision: Any, source_classes: Any) -> NDArray[np.float64]: z = np.asarray(decision, dtype=np.float64) classes = np.asarray(source_classes, dtype=np.int64) if z.ndim == 1: if classes.size != 2: raise ValueError('A one-dimensional decision function is only valid for two classes') z = np.stack([-z, z], axis=1) if z.ndim != 2 or z.shape[1] != classes.size: raise ValueError(f'Decision-function shape {z.shape} does not match source classes {classes.tolist()}') z = z - np.max(z, axis=1, keepdims=True) e = np.exp(z) return e / np.maximum(e.sum(axis=1, keepdims=True), 1e-12) def align_proba(proba: Any, source_classes: Any, target_classes: Any, *, normalize: bool=True) -> NDArray[np.float64]: p = np.asarray(proba, dtype=np.float64) src = np.asarray(source_classes, dtype=np.int64) dst = np.asarray(target_classes, dtype=np.int64) if p.ndim != 2: raise ValueError(f'Expected a 2-D probability matrix, got {p.shape}') if p.shape[1] != src.size: raise ValueError(f'Probability width {p.shape[1]} != number of source classes {src.size}') if len(np.unique(src)) != src.size or len(np.unique(dst)) != dst.size: raise ValueError('Class IDs must be unique') lookup = {int(c): j for j, c in enumerate(dst)} out = np.zeros((p.shape[0], dst.size), dtype=np.float64) for j, c in enumerate(src): if int(c) not in lookup: raise ValueError(f'Estimator emitted unknown class ID {int(c)}') out[:, lookup[int(c)]] = p[:, j] out = np.clip(out, 0.0, np.inf) if normalize: denom = out.sum(axis=1, keepdims=True) zero = denom[:, 0] <= 0 if np.any(zero): out[zero] = 1.0 / max(dst.size, 1) denom = out.sum(axis=1, keepdims=True) out /= denom return out def estimator_proba(estimator: Any, features: Any, target_classes: Any) -> NDArray[np.float64]: src = estimator_classes(estimator) if hasattr(estimator, 'predict_proba'): local = np.asarray(estimator.predict_proba(features), dtype=np.float64) elif hasattr(estimator, 'decision_function'): local = decision_to_proba(estimator.decision_function(features), src) else: pred = np.asarray(estimator.predict(features), dtype=np.int64) local = np.zeros((pred.size, src.size), dtype=np.float64) src_lookup = {int(c): i for i, c in enumerate(src)} for i, c in enumerate(pred): local[i, src_lookup[int(c)]] = 1.0 return align_proba(local, src, target_classes) def labels_from_proba(proba: Any, target_classes: Any) -> NDArray[np.int64]: p = np.asarray(proba, dtype=np.float64) classes = np.asarray(target_classes, dtype=np.int64) return classes[np.argmax(p, axis=1)].astype(np.int64)