sphncs 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sphncs/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """Similarity-Preserving Hierarchical Nonparametric Clustering System."""
2
+
3
+ from .estimator import SphncsClusterer
4
+ from .logs import LogSPHNCS
5
+ from .persistence import ModelPersistenceError
6
+ from .preprocessing import LogPreprocessor
7
+
8
+ __all__ = ["LogPreprocessor", "LogSPHNCS", "ModelPersistenceError", "SphncsClusterer"]
sphncs/density.py ADDED
@@ -0,0 +1,81 @@
1
+ """One-dimensional KDE evaluation, extrema, labels, and representatives."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+ import numpy as np
8
+ from scipy.signal import find_peaks
9
+
10
+
11
+ @dataclass
12
+ class DensityModel:
13
+ grid: np.ndarray
14
+ density: np.ndarray
15
+ boundaries: np.ndarray
16
+ modes: np.ndarray
17
+ labels: np.ndarray
18
+ representative_indices: np.ndarray
19
+
20
+ def predict(self, values: np.ndarray) -> np.ndarray:
21
+ return np.searchsorted(self.boundaries, values, side="right").astype(int)
22
+
23
+
24
+ def fit_density(
25
+ values: np.ndarray,
26
+ *,
27
+ bandwidth: str | float = "ISJ",
28
+ grid_points: int = 1024,
29
+ prominence: float | None = None,
30
+ prominence_fraction: float = 0.05,
31
+ min_samples: int = 3,
32
+ ) -> DensityModel:
33
+ """Fit FFTKDE and turn meaningful minima into interval labels."""
34
+ values = np.asarray(values, dtype=float).reshape(-1)
35
+ if values.size == 0:
36
+ raise ValueError("KDE requires at least one value")
37
+ lower, upper = float(values.min()), float(values.max())
38
+ if values.size < min_samples or np.isclose(lower, upper):
39
+ grid = np.linspace(lower - 0.5, upper + 0.5 if upper >= lower else lower + 0.5, max(2, grid_points))
40
+ density = np.zeros_like(grid)
41
+ mode = np.array([float(values.mean())])
42
+ labels = np.zeros(values.size, dtype=int)
43
+ return DensityModel(grid, density, np.array([], dtype=float), mode, labels, np.array([int(np.argmin(abs(values - mode[0])))]) )
44
+
45
+ padding = max((upper - lower) * 0.1, np.finfo(float).eps * 100)
46
+ grid = np.linspace(lower - padding, upper + padding, grid_points)
47
+ try:
48
+ from KDEpy import FFTKDE
49
+ except ImportError as exc: # pragma: no cover - dependency declaration is authoritative
50
+ raise ImportError("KDEpy is required; install sphncs with its runtime dependencies.") from exc
51
+ try:
52
+ density = np.asarray(FFTKDE(kernel="gaussian", bw=bandwidth).fit(values).evaluate(grid), dtype=float)
53
+ except ValueError:
54
+ # ISJ can fail to find a root for small, highly discrete partitions.
55
+ # Keep an explicit user-selected bandwidth strict, but make the default
56
+ # automatic choice robust by falling back to another KDEpy rule.
57
+ if not isinstance(bandwidth, str) or bandwidth.upper() != "ISJ":
58
+ raise
59
+ density = np.asarray(FFTKDE(kernel="gaussian", bw="silverman").fit(values).evaluate(grid), dtype=float)
60
+ scale = float(np.ptp(density))
61
+ effective_prominence = (
62
+ prominence
63
+ if prominence is not None
64
+ else max(scale * prominence_fraction, np.finfo(float).eps)
65
+ )
66
+ minima, _ = find_peaks(-density, prominence=effective_prominence)
67
+ boundaries = grid[minima]
68
+ labels = np.searchsorted(boundaries, values, side="right").astype(int)
69
+
70
+ modes: list[float] = []
71
+ representatives: list[int] = []
72
+ for label in range(len(boundaries) + 1):
73
+ left = -np.inf if label == 0 else boundaries[label - 1]
74
+ right = np.inf if label == len(boundaries) else boundaries[label]
75
+ mask = (grid > left) & (grid <= right)
76
+ mode = float(grid[mask][np.argmax(density[mask])])
77
+ members = np.flatnonzero(labels == label)
78
+ if members.size:
79
+ modes.append(mode)
80
+ representatives.append(int(members[np.argmin(abs(values[members] - mode))]))
81
+ return DensityModel(grid, density, boundaries, np.asarray(modes), labels, np.asarray(representatives, dtype=int))
sphncs/distances.py ADDED
@@ -0,0 +1,99 @@
1
+ """Reusable distance metrics for SPHNCS object spaces."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections import Counter
6
+ from collections.abc import Callable, Iterable
7
+ from functools import lru_cache
8
+ from math import inf, isfinite
9
+ from numbers import Real
10
+
11
+ Metric = Callable[[object, object], float]
12
+ StringMetric = Callable[[str, str], float]
13
+
14
+
15
+ def lp_distance(left: Iterable[Real], right: Iterable[Real], p: Real = 2) -> float:
16
+ """Return Minkowski Lp distance between two equally sized numeric vectors.
17
+
18
+ ``p`` must be at least one. ``p=float('inf')`` returns the Chebyshev
19
+ (L-infinity) distance. Inputs are materialized once so any finite iterable
20
+ of real values is accepted.
21
+ """
22
+ if isinstance(p, bool) or not isinstance(p, Real) or p < 1:
23
+ raise ValueError("p must be a real number greater than or equal to 1")
24
+ left_values = tuple(left)
25
+ right_values = tuple(right)
26
+ if len(left_values) != len(right_values):
27
+ raise ValueError("vectors must have equal lengths")
28
+ try:
29
+ differences = [abs(float(a) - float(b)) for a, b in zip(left_values, right_values)]
30
+ except (TypeError, ValueError) as exc:
31
+ raise TypeError("vectors must contain real numeric values") from exc
32
+ if not all(isfinite(value) for value in differences):
33
+ raise ValueError("vectors must contain finite numeric values")
34
+ if p == inf:
35
+ return max(differences, default=0.0)
36
+ return float(sum(value**p for value in differences) ** (1 / p))
37
+
38
+
39
+ def l1_distance(left: Iterable[Real], right: Iterable[Real]) -> float:
40
+ """Return Manhattan (L1) distance between numeric vectors."""
41
+ return lp_distance(left, right, p=1)
42
+
43
+
44
+ def l2_distance(left: Iterable[Real], right: Iterable[Real]) -> float:
45
+ """Return Euclidean (L2) distance between numeric vectors."""
46
+ return lp_distance(left, right, p=2)
47
+
48
+
49
+ @lru_cache(maxsize=32_768)
50
+ def _char_ngram_counts(value: str, ngram_size: int) -> Counter[str]:
51
+ """Build a multiset of character shingles once per immutable input string."""
52
+ return Counter(value[i : i + ngram_size] for i in range(max(1, len(value) - ngram_size + 1)))
53
+
54
+
55
+ def normalized_levenshtein(left: str, right: str) -> float:
56
+ """Levenshtein edit distance normalized to the interval [0, 1]."""
57
+ if left == right:
58
+ return 0.0
59
+ if not left or not right:
60
+ return 1.0
61
+ if len(left) < len(right):
62
+ left, right = right, left
63
+ previous = list(range(len(right) + 1))
64
+ for i, char_left in enumerate(left, start=1):
65
+ current = [i]
66
+ for j, char_right in enumerate(right, start=1):
67
+ current.append(min(current[-1] + 1, previous[j] + 1, previous[j - 1] + (char_left != char_right)))
68
+ previous = current
69
+ return previous[-1] / len(left)
70
+
71
+
72
+ def char_ngram_jaccard(left: str, right: str, ngram_size: int = 3) -> float:
73
+ """Multiset character n-gram Jaccard distance."""
74
+ if ngram_size < 1:
75
+ raise ValueError("ngram_size must be positive")
76
+ if left == right:
77
+ return 0.0
78
+ grams_left = _char_ngram_counts(left, ngram_size)
79
+ grams_right = _char_ngram_counts(right, ngram_size)
80
+ union = sum((grams_left | grams_right).values())
81
+ return 1.0 - sum((grams_left & grams_right).values()) / union if union else 0.0
82
+
83
+
84
+ METRICS: dict[str, Metric] = {
85
+ "l1": l1_distance,
86
+ "l2": l2_distance,
87
+ "normalized_levenshtein": normalized_levenshtein,
88
+ "char_ngram_jaccard": char_ngram_jaccard,
89
+ }
90
+
91
+
92
+ def resolve_metric(metric: str | Metric) -> Metric:
93
+ if callable(metric):
94
+ return metric
95
+ try:
96
+ return METRICS[metric]
97
+ except KeyError as exc:
98
+ choices = ", ".join(sorted(METRICS))
99
+ raise ValueError(f"Unknown metric {metric!r}; choose one of {choices}, or pass a callable.") from exc
sphncs/embedding.py ADDED
@@ -0,0 +1,136 @@
1
+ """Adapter around fastmapy's batched, on-demand FastMap implementation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import random
6
+ import sys
7
+ from collections.abc import Callable
8
+ from pathlib import Path
9
+
10
+ import numpy as np
11
+
12
+
13
+ def _load_fastmap():
14
+ """Import the installed package or the checked-out project submodule."""
15
+ try:
16
+ from fastmap import FastMap
17
+ return FastMap
18
+ except ImportError:
19
+ submodule = Path(__file__).resolve().parent.parent / "vendor" / "fastmapy"
20
+ if submodule.is_dir():
21
+ sys.path.insert(0, str(submodule))
22
+ try:
23
+ from fastmap import FastMap
24
+ return FastMap
25
+ except ImportError:
26
+ pass
27
+ raise ImportError(
28
+ "fastmapy is required. Install FastMapy or initialize and install the "
29
+ "project submodule with `git submodule update --init --recursive` and "
30
+ "`python -m pip install -e vendor/fastmapy`."
31
+ )
32
+
33
+
34
+ class _CallableDistance:
35
+ """FastMapy-compatible, pickleable wrapper for an SPHNCS metric callable."""
36
+
37
+ def __init__(self, metric: Callable[[object, object], float]):
38
+ self.metric = metric
39
+
40
+ @staticmethod
41
+ def get_name() -> str:
42
+ return "sphncs_metric"
43
+
44
+ def calculate(self, left: object, right: object) -> float:
45
+ value = float(self.metric(left, right))
46
+ if value < 0:
47
+ raise ValueError("Metrics must return non-negative distances")
48
+ return value
49
+
50
+
51
+ class FastMapyEmbeddings:
52
+ """A batch of independent one-dimensional fastmapy models.
53
+
54
+ Each model gets a distinct pivot pair through ``FastMap.fit_many``. This
55
+ intentionally differs from a single multi-dimensional FastMap embedding:
56
+ sphncs uses every independent 1-D density partition as consensus evidence.
57
+ """
58
+
59
+ def __init__(
60
+ self, n_embeddings: int, metric: Callable[[object, object], float], *,
61
+ random_state: int | None = None, cores: int = 1, iters: int = 3,
62
+ cache_distances: bool = True,
63
+ ):
64
+ if n_embeddings < 1:
65
+ raise ValueError("n_embeddings must be at least 1")
66
+ self.n_embeddings = n_embeddings
67
+ self.metric = metric
68
+ self.random_state = random_state
69
+ self.cores = cores
70
+ self.iters = iters
71
+ self.cache_distances = cache_distances
72
+
73
+ def fit(self, X: list[object]):
74
+ if not X:
75
+ raise ValueError("FastMap requires at least one string")
76
+ self.X_ = list(X)
77
+ if len(X) == 1:
78
+ self.models_ = []
79
+ self.coordinates_ = np.zeros((1, self.n_embeddings), dtype=float)
80
+ return self
81
+ FastMap = _load_fastmap()
82
+
83
+ count = min(self.n_embeddings, len(X))
84
+ # fastmapy currently draws its pivot starts from Python's module-level
85
+ # RNG. Restore it afterwards so fitting sphncs does not perturb callers.
86
+ rng_state = random.getstate()
87
+ try:
88
+ if self.random_state is not None:
89
+ random.seed(self.random_state)
90
+ self.models_ = FastMap.fit_many(
91
+ X,
92
+ count=count,
93
+ dim=1,
94
+ distance=_CallableDistance,
95
+ dist_args={"metric": self.metric},
96
+ cores=self.cores,
97
+ iters=self.iters,
98
+ )
99
+ finally:
100
+ random.setstate(rng_state)
101
+ coordinates = np.column_stack([np.asarray(model.transform(X), dtype=float).reshape(-1) for model in self.models_])
102
+ if count < self.n_embeddings:
103
+ coordinates = np.pad(coordinates, ((0, 0), (0, self.n_embeddings - count)))
104
+ self.coordinates_ = coordinates
105
+ return self
106
+
107
+ def fit_transform(self, X: list[object]) -> np.ndarray:
108
+ return self.fit(X).coordinates_.copy()
109
+
110
+ def transform(self, X: list[object]) -> np.ndarray:
111
+ if not hasattr(self, "models_"):
112
+ raise RuntimeError("FastMap must be fitted before transform")
113
+ if not self.models_:
114
+ return np.zeros((len(X), self.n_embeddings), dtype=float)
115
+ coordinates = np.column_stack([np.asarray(model.transform(X), dtype=float).reshape(-1) for model in self.models_])
116
+ if coordinates.shape[1] < self.n_embeddings:
117
+ coordinates = np.pad(coordinates, ((0, 0), (0, self.n_embeddings - coordinates.shape[1])))
118
+ return coordinates
119
+
120
+
121
+ def save_models(self, directory: Path) -> list[Path]:
122
+ """Persist each fitted projection through FastMapy's native format."""
123
+ if not hasattr(self, "models_"):
124
+ raise RuntimeError("FastMap must be fitted before it can be saved")
125
+ directory.mkdir(parents=True, exist_ok=True)
126
+ paths: list[Path] = []
127
+ for index, model in enumerate(self.models_):
128
+ path = directory / f"{index}.fastmap"
129
+ model.save(path)
130
+ paths.append(path)
131
+ return paths
132
+
133
+ def load_models(self, paths: list[Path]) -> None:
134
+ """Restore fitted projections through FastMapy's native loader."""
135
+ FastMap = _load_fastmap()
136
+ self.models_ = [FastMap.load(path) for path in paths]