aether-context 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,213 @@
1
+ # aether-context (Unlimited Context)
2
+ # Copyright (c) 2026 Aether AI
3
+ # SPDX-License-Identifier: Apache-2.0
4
+ """B1 static encoder — Model2Vec-style, numpy-only, 256-dim, stateless.
5
+
6
+ Maps text to a fixed-size retrieval embedding the rest of the engine indexes and
7
+ searches over. The pipeline is the whole of it:
8
+
9
+ text -> regex tokenize (lowercased word tokens)
10
+ -> per-token row (hashed token -> seeded RNG -> deterministic unit row)
11
+ -> mean-pool the rows
12
+ -> L2-normalize
13
+ -> 256-dim float32 unit vector
14
+
15
+ There is **no shipped multi-hundred-MB asset**: each token's row is *generated on
16
+ demand* from a hash of the token, seeded so the same token always yields the same
17
+ row (within a process and across processes / instances). Because shared tokens map
18
+ to the same row, mean-pooling gives real lexical cosine structure — two strings that
19
+ share tokens land closer than two that share none. Distinct tokens draw
20
+ near-orthogonal rows in 256-d, so unrelated text sits near cosine 0.
21
+
22
+ This produces the **256-dim retrieval embedding only** — it is purely a retrieval key
23
+ and not an attention mechanism. Swapping in a trained Model2Vec table later is a drop-in
24
+ replacement for ``_row_for_token`` / the lookup — the public API stays fixed.
25
+
26
+ Stateless and shared: a single ``StaticEncoder`` instance is safe to reuse; it holds
27
+ only a tiny in-memory cache of generated rows for speed.
28
+ """
29
+ from __future__ import annotations
30
+
31
+ import hashlib
32
+ import re
33
+ from typing import Iterable
34
+
35
+ import numpy as np
36
+
37
+ from aether_context._log import get_logger
38
+ from aether_context.errors import EncoderError
39
+
40
+ logger = get_logger(__name__)
41
+
42
+ #: Version pin for the encoding scheme. Bump when the token->row generation, the
43
+ #: pooling, or the normalization changes (vectors produced by different versions are
44
+ #: not comparable). Retrieval indexes should record this alongside stored vectors.
45
+ ENCODER_VERSION: str = "static_v1"
46
+
47
+ #: Default embedding dimensionality (the 256-dim retrieval embedding).
48
+ DEFAULT_DIM: int = 256
49
+
50
+ #: Master seed mixed into every per-token hash so the whole table is reproducible but
51
+ #: namespaced to this encoder version.
52
+ _SEED_NAMESPACE: str = f"aether_context.encoder/{ENCODER_VERSION}"
53
+
54
+ #: Word tokenizer: runs of letters/digits/underscore. Punctuation and whitespace are
55
+ #: separators (so "refactor, the" and "refactor the" tokenize identically).
56
+ _TOKEN_RE = re.compile(r"[a-z0-9_]+")
57
+
58
+
59
+ def _tokenize(text: str) -> list[str]:
60
+ """Lowercase and split ``text`` into word tokens.
61
+
62
+ Case-insensitive (lowercased first) and punctuation-insensitive (punctuation is a
63
+ separator, never part of a token). Returns an empty list for empty / whitespace /
64
+ punctuation-only input.
65
+ """
66
+ return _TOKEN_RE.findall(text.lower())
67
+
68
+
69
+ class StaticEncoder:
70
+ """Deterministic, stateless, numpy-only text -> unit-vector encoder.
71
+
72
+ Args:
73
+ dim: Output dimensionality. Defaults to :data:`DEFAULT_DIM` (256). Must be a
74
+ positive integer.
75
+
76
+ The encoder is immutable after construction (only an internal row cache mutates,
77
+ which is a pure performance detail and never affects output). Reuse one instance.
78
+ """
79
+
80
+ def __init__(self, dim: int = DEFAULT_DIM) -> None:
81
+ if not isinstance(dim, int) or isinstance(dim, bool) or dim <= 0:
82
+ raise EncoderError(
83
+ f"StaticEncoder dim must be a positive int, got {dim!r}.",
84
+ hint="Pass dim=256 (default) or another positive integer.",
85
+ )
86
+ self._dim: int = dim
87
+ # token -> generated unit row; pure speed cache, never affects determinism.
88
+ self._row_cache: dict[str, np.ndarray] = {}
89
+ # A normalized empty/degenerate fallback so callers always get a unit vector.
90
+ self._empty_vector: np.ndarray = self._make_empty_vector()
91
+
92
+ @property
93
+ def dim(self) -> int:
94
+ """The output embedding dimensionality."""
95
+ return self._dim
96
+
97
+ @property
98
+ def version(self) -> str:
99
+ """The pinned encoder version string (:data:`ENCODER_VERSION`)."""
100
+ return ENCODER_VERSION
101
+
102
+ # -- public API ----------------------------------------------------------
103
+ def encode(self, text: str) -> np.ndarray:
104
+ """Encode one string into a ``(dim,)`` float32 unit vector.
105
+
106
+ Empty / whitespace-only / punctuation-only input yields a deterministic unit
107
+ fallback vector (never raises, never returns NaN). Non-string input raises
108
+ :class:`~aether_context.errors.EncoderError`.
109
+ """
110
+ if not isinstance(text, str):
111
+ raise EncoderError(
112
+ f"encode() expects str, got {type(text).__name__}.",
113
+ hint="Pass the text as a Python str (decode bytes first).",
114
+ )
115
+ tokens = _tokenize(text)
116
+ if not tokens:
117
+ return self._empty_vector.copy()
118
+ rows = np.empty((len(tokens), self._dim), dtype=np.float32)
119
+ for i, tok in enumerate(tokens):
120
+ rows[i] = self._row_for_token(tok)
121
+ pooled = rows.mean(axis=0)
122
+ return self._l2_normalize(pooled)
123
+
124
+ def encode_batch(self, texts: Iterable[str]) -> np.ndarray:
125
+ """Encode many strings into an ``(N, dim)`` float32 matrix of unit rows.
126
+
127
+ An empty iterable yields a ``(0, dim)`` array. Each row is produced exactly as
128
+ :meth:`encode` would produce it, so ``encode_batch`` and per-item ``encode``
129
+ agree. Non-iterable input raises
130
+ :class:`~aether_context.errors.EncoderError`.
131
+ """
132
+ try:
133
+ items = list(texts)
134
+ except TypeError as exc:
135
+ raise EncoderError(
136
+ f"encode_batch() expects an iterable of str, got "
137
+ f"{type(texts).__name__}.",
138
+ hint="Pass a list/tuple of strings, e.g. encode_batch([\"a\", \"b\"]).",
139
+ ) from exc
140
+ if not items:
141
+ return np.empty((0, self._dim), dtype=np.float32)
142
+ out = np.empty((len(items), self._dim), dtype=np.float32)
143
+ for i, text in enumerate(items):
144
+ out[i] = self.encode(text)
145
+ return out
146
+
147
+ # -- internals -----------------------------------------------------------
148
+ def _row_for_token(self, token: str) -> np.ndarray:
149
+ """Return the deterministic unit row for ``token``, generating + caching it.
150
+
151
+ The row is drawn from a per-token-seeded RNG (Gaussian) and L2-normalized, so
152
+ each distinct token maps to a fixed unit vector that is near-orthogonal to
153
+ other tokens' rows in high dimension. Shared tokens => identical rows => real
154
+ lexical cosine structure after mean-pooling.
155
+ """
156
+ cached = self._row_cache.get(token)
157
+ if cached is not None:
158
+ return cached
159
+ seed = self._seed_for_token(token)
160
+ rng = np.random.default_rng(seed)
161
+ row = rng.standard_normal(self._dim).astype(np.float32)
162
+ row = self._l2_normalize(row)
163
+ self._row_cache[token] = row
164
+ return row
165
+
166
+ @staticmethod
167
+ def _seed_for_token(token: str) -> int:
168
+ """Deterministic 64-bit seed for a token (stable across processes / hosts).
169
+
170
+ Uses BLAKE2b over a versioned namespace + the token so the table is
171
+ reproducible everywhere and tied to :data:`ENCODER_VERSION`. (Not security
172
+ sensitive — purely a stable hash -> seed.)
173
+ """
174
+ digest = hashlib.blake2b(
175
+ f"{_SEED_NAMESPACE}\x00{token}".encode("utf-8"), digest_size=8
176
+ ).digest()
177
+ return int.from_bytes(digest, "big", signed=False)
178
+
179
+ def _make_empty_vector(self) -> np.ndarray:
180
+ """Deterministic unit fallback for empty/degenerate input.
181
+
182
+ Generated from a reserved sentinel token so it is a valid unit vector (callers
183
+ can always assume unit norm) yet distinct from any real single-token vector.
184
+ """
185
+ seed = self._seed_for_token("\x00<empty>\x00")
186
+ rng = np.random.default_rng(seed)
187
+ row = rng.standard_normal(self._dim).astype(np.float32)
188
+ return self._l2_normalize_raw(row)
189
+
190
+ def _l2_normalize(self, vec: np.ndarray) -> np.ndarray:
191
+ """L2-normalize a vector to unit length as float32.
192
+
193
+ A zero / degenerate vector (norm too small to divide safely) falls back to the
194
+ deterministic empty vector so the result is always a finite unit vector.
195
+ """
196
+ norm = float(np.linalg.norm(vec))
197
+ if norm < 1e-12:
198
+ return self._empty_vector.copy()
199
+ return (vec / norm).astype(np.float32)
200
+
201
+ @staticmethod
202
+ def _l2_normalize_raw(vec: np.ndarray) -> np.ndarray:
203
+ """L2-normalize without the empty-vector fallback (used to build it)."""
204
+ norm = float(np.linalg.norm(vec))
205
+ if norm < 1e-12:
206
+ # Astronomically unlikely for a Gaussian draw; degrade to a fixed axis.
207
+ out = np.zeros_like(vec, dtype=np.float32)
208
+ out[0] = 1.0
209
+ return out
210
+ return (vec / norm).astype(np.float32)
211
+
212
+
213
+ __all__ = ["StaticEncoder", "ENCODER_VERSION", "DEFAULT_DIM"]
@@ -0,0 +1,93 @@
1
+ # aether-context (Unlimited Context)
2
+ # Copyright (c) 2026 Aether AI
3
+ # SPDX-License-Identifier: Apache-2.0
4
+ """Typed errors for aether-context.
5
+
6
+ Design intent: every failure in the library re-wraps the underlying cause into one of
7
+ these typed errors. No module in the package may use a bare ``except:`` — catch the
8
+ specific exception and re-raise as one of these, so callers always get an actionable
9
+ ``.hint`` with the exact fix command.
10
+
11
+ Fail-soft discipline lives in the call sites (the pager/retrieval degrades and logs
12
+ rather than raising into a long run); these types exist so that when something *is*
13
+ worth surfacing, it surfaces with a fix.
14
+ """
15
+ from __future__ import annotations
16
+
17
+
18
+ class AetherContextError(Exception):
19
+ """Base class for all aether-context errors.
20
+
21
+ Carries a human-actionable ``.hint`` describing how to fix the condition. Subclasses
22
+ provide a sensible default hint; callers may override it per-instance.
23
+ """
24
+
25
+ #: Default fix hint; subclasses override. Always a non-empty string.
26
+ default_hint: str = "See the docs at https://github.com/aethersystems/unlimited-context"
27
+
28
+ def __init__(self, message: str, *, hint: str | None = None) -> None:
29
+ super().__init__(message)
30
+ self.message: str = message
31
+ self.hint: str = hint if hint is not None else self.default_hint
32
+
33
+ def __str__(self) -> str:
34
+ return f"{self.message}\nhint: {self.hint}"
35
+
36
+
37
+ class PoolBudgetError(AetherContextError):
38
+ """The context pool exceeded (or cannot satisfy) its GB budget ceiling."""
39
+
40
+ default_hint = (
41
+ "Raise the pool size (e.g. `aether-context --pool 10`) or let the witness evict "
42
+ "stale slices; the pool budget floor is 5 GB."
43
+ )
44
+
45
+
46
+ class OllamaNotRunning(AetherContextError):
47
+ """The Ollama daemon is not reachable on localhost:11434."""
48
+
49
+ default_hint = "Start the Ollama daemon: `ollama serve`"
50
+
51
+
52
+ class ModelNotPulled(AetherContextError):
53
+ """The requested Ollama model has not been downloaded yet."""
54
+
55
+ default_hint = "Pull the model first: `ollama pull <model>` (or pass `pull=True`)"
56
+
57
+
58
+ class BackendUnavailable(AetherContextError):
59
+ """A backend's optional dependency is missing or the backend cannot be loaded."""
60
+
61
+ default_hint = (
62
+ "Install the backend extra, e.g. `pip install \"aether-context[llamacpp]\"` or "
63
+ "`pip install \"aether-context[hf]\"`; or use `model=\"mock\"` to run offline."
64
+ )
65
+
66
+
67
+ class EncoderError(AetherContextError):
68
+ """The static encoder failed to produce an embedding."""
69
+
70
+ default_hint = (
71
+ "Check the input text is a non-empty string; if a custom encoder table was "
72
+ "provided, verify its shape is (vocab, dim)."
73
+ )
74
+
75
+
76
+ class PoolCorrupt(AetherContextError):
77
+ """The on-disk context pool (mmap index or payloads) is malformed."""
78
+
79
+ default_hint = (
80
+ "Re-initialize the pool with `aether-context init`, or delete the pool dir "
81
+ "under ~/.aether-context to rebuild it (non-destructive to your models)."
82
+ )
83
+
84
+
85
+ __all__ = [
86
+ "AetherContextError",
87
+ "PoolBudgetError",
88
+ "OllamaNotRunning",
89
+ "ModelNotPulled",
90
+ "BackendUnavailable",
91
+ "EncoderError",
92
+ "PoolCorrupt",
93
+ ]